{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,7]],"date-time":"2026-07-07T04:36:09Z","timestamp":1783398969374,"version":"3.54.6"},"reference-count":46,"publisher":"IEEE","license":[{"start":{"date-parts":[[2026,1,31]],"date-time":"2026-01-31T00:00:00Z","timestamp":1769817600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,1,31]],"date-time":"2026-01-31T00:00:00Z","timestamp":1769817600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026,1,31]]},"DOI":"10.1109\/hpca68181.2026.11408552","type":"proceedings-article","created":{"date-parts":[[2026,3,4]],"date-time":"2026-03-04T20:47:22Z","timestamp":1772657242000},"page":"1-15","source":"Crossref","is-referenced-by-count":1,"title":["VectorLiteRAG: Latency-Aware and Fine-Grained Resource Partitioning for Efficient RAG"],"prefix":"10.1109","author":[{"given":"Junkyum","family":"Kim","sequence":"first","affiliation":[{"name":"Georgia Institute of Technology"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Divya","family":"Mahajan","sequence":"additional","affiliation":[{"name":"Georgia Institute of Technology"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.14778\/3485450.3485462"},{"key":"ref2","volume-title":"Ad-rec: Advanced feature interactions to address covariate-shifts in recommendation networks","author":"Adnan","year":"2023"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA59077.2024.00081"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.14778\/2856318.2856324"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1145\/3340531.3412779"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/tbdata.2025.3618474"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1145\/3637528.3671470"},{"key":"ref8","article-title":"Retrieval-augmented generation for large language models: A survey","author":"Gao","year":"2023","journal-title":"arXiv preprint"},{"key":"ref9","volume-title":"The llama 3 herd of models","author":"Grattafiori","year":"2024"},{"key":"ref10","first-page":"3929","article-title":"Retrieval augmented language model pre-training","volume-title":"International conference on machine learning. PMLR","author":"Guu","year":"2020"},{"key":"ref11","article-title":"Hedrarag: Coordinating 11 m generation and database retrieval in heterogeneous rag serving","author":"Hu","year":"2025","journal-title":"arXiv preprint"},{"key":"ref12","first-page":"585","article-title":"Cxl-anns:software-hardware collaborative memory disaggregation and computation for billion-scale approximate nearest neighbor search","volume-title":"2023 USENIX Annual Technical Conference (USENIX ATC 23)","author":"Jang","year":"2023"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2010.57"},{"key":"ref14","article-title":"Chameleon: a heterogeneous and disaggregated accelerator system for retrievalaugmented language models","author":"Jiang","year":"2023","journal-title":"arXiv preprint"},{"key":"ref15","article-title":"Piperag: Fast retrieval-augmented generation via algorithm-system co-design","author":"Jiang","year":"2024","journal-title":"arXiv preprint"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1145\/3768628"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/TBDATA.2019.2921572"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1145\/3600006.3613165"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1145\/3470496.3527386"},{"key":"ref20","article-title":"Langchain: Context-aware reasoning framework","year":"2025","journal-title":"LangChain-Team"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA53966.2022.00021"},{"key":"ref22","article-title":"Retrieval-augmented generation for knowledge-intensive nlp tasks","volume-title":"Proceedings of the 34th International Conference on Neural Information Processing Systems","author":"Lewis"},{"key":"ref23","volume-title":"Telerag: Efficient retrieval-augmented generation inference with lookahead retrieval","author":"Lin","year":"2025"},{"key":"ref24","volume-title":"Llamaindex","author":"Liu","year":"2022"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.23919\/DATE64628.2025.10992746"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1145\/3695053.3731032"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2018.2889473"},{"key":"ref28","article-title":"Pointer sentinel mixture models","author":"Merity","year":"2016","journal-title":"arXiv preprint"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1145\/3470496.3533727"},{"key":"ref30","volume-title":"Text and code embeddings by contrastive pre-training","author":"Neelakantan","year":"2022"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA59077.2024.00019"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1145\/3669940.3707264"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1162\/tacl_a_00605"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/d19-1410"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1145\/3695053.3731076"},{"key":"ref36","first-page":"16857","article-title":"Mpnet: Masked and permuted pre-training for language understanding","volume":"33","author":"Song","year":"2020","journal-title":"Advances in neural information processing systems"},{"key":"ref37","author":"Team","year":"2024","journal-title":"Wiki-all dataset"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1145\/3448016.3457550"},{"key":"ref39","first-page":"25","article-title":"A theoretical analysis of ndcg type ranking measures","volume-title":"Conference on learning theory. PMLR","author":"Wang","year":"2013"},{"key":"ref40","article-title":"Wikipedia dumps","year":"2025","journal-title":"Wikimedia Foundation"},{"key":"ref41","volume-title":"Qwen3 technical report","author":"Yang","year":"2025"},{"key":"ref42","article-title":"Jasper and stella: distillation of sota embedding models","author":"Zhang","year":"2024","journal-title":"arXiv preprint"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v36i4.20356"},{"key":"ref44","article-title":"Accelerating retrieval-augmented language model serving with speculation","author":"Zhang","year":"2024","journal-title":"arXiv preprint"},{"key":"ref45","first-page":"193","article-title":"Distserve: Disaggregating prefill and decoding for goodputoptimized large language model serving","volume-title":"18th USENIX Symposium on Operating Systems Design and Implementation (OSDI 24)","author":"Zhong","year":"2024"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.1145\/1132956.1132959"}],"event":{"name":"2026 IEEE International Symposium on High Performance Computer Architecture (HPCA)","location":"Sydney, Australia","start":{"date-parts":[[2026,1,31]]},"end":{"date-parts":[[2026,2,4]]}},"container-title":["2026 IEEE International Symposium on High Performance Computer Architecture (HPCA)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11408404\/11408433\/11408552.pdf?arnumber=11408552","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,3,5]],"date-time":"2026-03-05T06:35:58Z","timestamp":1772692558000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11408552\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,1,31]]},"references-count":46,"URL":"https:\/\/doi.org\/10.1109\/hpca68181.2026.11408552","relation":{},"subject":[],"published":{"date-parts":[[2026,1,31]]}}}