{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,18]],"date-time":"2026-06-18T14:55:39Z","timestamp":1781794539264,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":43,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,6,22]],"date-time":"2026-06-22T00:00:00Z","timestamp":1782086400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,6,22]]},"DOI":"10.1145\/3787109.3815268","type":"proceedings-article","created":{"date-parts":[[2026,6,18]],"date-time":"2026-06-18T14:17:19Z","timestamp":1781792239000},"page":"743-749","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["HeteroRAGCache: Software-Hardware Co-Design for Efficient RAG Caching using Emerging Memories"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0005-3508-3859","authenticated-orcid":false,"given":"Jangseon","family":"Park","sequence":"first","affiliation":[{"name":"Computer Science and Engineering, University of California San Diego, San Diego, CA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-0198-2613","authenticated-orcid":false,"given":"Kiseok","family":"Suh","sequence":"additional","affiliation":[{"name":"Foundry Business Division, Samsung Electronics, Suwon, gyeonggi-do, Republic of Korea"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9662-498X","authenticated-orcid":false,"given":"Flavio","family":"Ponzina","sequence":"additional","affiliation":[{"name":"Electrical and Computer Engineering, San Diego State University, San Diego, CA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6954-997X","authenticated-orcid":false,"given":"Tajana","family":"Rosing","sequence":"additional","affiliation":[{"name":"Computer Science and Engineering, University of California San Diego, San Diego, CA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,6,22]]},"reference":[{"key":"e_1_3_3_1_2_2","unstructured":"Meta AI. 2024. Meta-Llama-3-8B. https:\/\/huggingface.co\/meta-llama\/Meta-Llama-3-8B. Accessed: 2025-03-26."},{"key":"e_1_3_3_1_3_2","unstructured":"Abiola Ayodele. 2025. High Bandwidth Memory: Concepts Architecture and Applications. https:\/\/www.wevolver.com\/article\/high-bandwidth-memory. Accessed: 2025-03-27."},{"key":"e_1_3_3_1_4_2","volume-title":"The Twelfth International Conference on Learning Representations","author":"Chen Yukang","year":"2024","unstructured":"Yukang Chen, Shengju Qian, Haotian Tang, Xin Lai, Zhijian Liu, Song Han, and Jiaya Jia. 2024. LongLoRA: Efficient Fine-tuning of Long-Context Large Language Models. In The Twelfth International Conference on Learning Representations. OpenReview.net, Vienna, Austria."},{"key":"e_1_3_3_1_5_2","unstructured":"Aakanksha Chowdhery Sharan Narang Jacob Devlin Maarten Bosma Gaurav Mishra Adam Roberts Paul Barham Hyung\u00a0Won Chung Charles Sutton Sebastian Gehrmann et\u00a0al. 2023. Palm: Scaling language modeling with pathways. Journal of Machine Learning Research 24 240 (2023) 1\u2013113."},{"key":"e_1_3_3_1_6_2","doi-asserted-by":"crossref","unstructured":"Yulei Dai and Yanfeng Jiang. 2024. Design of Hierarchical Cache with Hybrid SOT-and STT-MRAM in Multi-core CPU Environments. IEEE Transactions on Magnetics 61 3 (2024) 1\u20138.","DOI":"10.1109\/TMAG.2024.3524579"},{"key":"e_1_3_3_1_7_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.naacl-main.365"},{"key":"e_1_3_3_1_8_2","doi-asserted-by":"crossref","unstructured":"Bernard Dieny Ioan\u00a0Lucian Prejbeanu Kevin Garello Pietro Gambardella Paulo Freitas Ronald Lehndorff Wolfgang Raberg Ursula Ebels Sergej\u00a0O Demokritov Johan Akerman et\u00a0al. 2020. Opportunities and challenges for spintronics in the microelectronics industry. Nature Electronics 3 8 (2020) 446\u2013459.","DOI":"10.1038\/s41928-020-0461-5"},{"key":"e_1_3_3_1_9_2","doi-asserted-by":"crossref","unstructured":"Yaosheng Fu Evgeny Bolotin Niladrish Chatterjee David Nellans and Stephen\u00a0W. Keckler. 2021. GPU Domain Specialization via Composable On-Package Architecture. ACM Transactions on Architecture and Code Optimization (TACO) 19 1 (2021) 4:1\u20134:23.","DOI":"10.1145\/3484505"},{"key":"e_1_3_3_1_10_2","unstructured":"IMEC. 2023. imec improves ferroelectric response and endurance of HZO-based ferroelectric capacitors. https:\/\/www.imec-int.com\/en\/press\/imec-improves-ferroelectric-response. Accessed: 2025-03-27."},{"key":"e_1_3_3_1_11_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.eacl-main.74"},{"key":"e_1_3_3_1_12_2","unstructured":"Gautier Izacard Patrick Lewis Maria Lomeli Lucas Hosseini Fabio Petroni Timo Schick Jane Dwivedi-Yu Armand Joulin Sebastian Riedel and Edouard Grave. 2023. Atlas: Few-shot learning with retrieval augmented language models. Journal of Machine Learning Research 24 251 (2023) 1\u201343."},{"key":"e_1_3_3_1_13_2","unstructured":"Wenqi Jiang Suvinay Subramanian Cat Graves Gustavo Alonso Amir Yazdanbakhsh and Vidushi Dadu. 2025. RAGO: Systematic Performance Optimization for Retrieval-Augmented Generation Serving. arxiv:https:\/\/arXiv.org\/abs\/2503.14649\u00a0[cs.IR] https:\/\/arxiv.org\/abs\/2503.14649"},{"key":"e_1_3_3_1_14_2","unstructured":"Wenqi Jiang Shuai Zhang Boran Han Jie Wang Bernie Wang and Tim Kraska. 2024. PipeRAG: Fast retrieval-augmented generation via algorithm-system co-design. arxiv:https:\/\/arXiv.org\/abs\/2403.05676\u00a0[cs.IR]"},{"key":"e_1_3_3_1_15_2","volume-title":"The Thirteenth International Conference on Learning Representations","author":"Jin Bowen","year":"2025","unstructured":"Bowen Jin, Jinsung Yoon, Jiawei Han, and Sercan\u00a0O Arik. 2025. Long-context llms meet rag: Overcoming challenges for long inputs in rag. In The Thirteenth International Conference on Learning Representations. OpenReview.net, Singapore."},{"key":"e_1_3_3_1_16_2","doi-asserted-by":"crossref","unstructured":"Chao Jin Zili Zhang Xuanlin Jiang Fangyue Liu Shufan Liu Xuanzhe Liu and Xin Jin. 2024. Ragcache: Efficient knowledge caching for retrieval-augmented generation. ACM Transactions on Computer Systems 44 1 (2024) 1\u201327.","DOI":"10.1145\/3768628"},{"key":"e_1_3_3_1_17_2","unstructured":"Jeff Johnson Matthijs Douze and Herv\u00e9 J\u00e9gou. 2024. Faiss: A Library for Efficient Similarity Search. https:\/\/github.com\/facebookresearch\/faiss."},{"key":"e_1_3_3_1_18_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.emnlp-main.550"},{"key":"e_1_3_3_1_19_2","doi-asserted-by":"crossref","unstructured":"Tom\u00e1\u0161 Ko\u010disk\u1ef3 Jonathan Schwarz Phil Blunsom Chris Dyer Karl\u00a0Moritz Hermann G\u00e1bor Melis and Edward Grefenstette. 2018. The narrativeqa reading comprehension challenge. Transactions of the Association for Computational Linguistics 6 (2018) 317\u2013328.","DOI":"10.1162\/tacl_a_00023"},{"key":"e_1_3_3_1_20_2","unstructured":"Patrick Lewis Ethan Perez Aleksandra Piktus Fabio Petroni Vladimir Karpukhin Naman Goyal Heinrich K\u00fcttler Mike Lewis Wen-tau Yih Tim Rockt\u00e4schel et\u00a0al. 2020. Retrieval-augmented generation for knowledge-intensive nlp tasks. Advances in Neural Information Processing Systems 33 (2020) 9459\u20139474."},{"key":"e_1_3_3_1_21_2","doi-asserted-by":"crossref","unstructured":"Yueting Li Xueyan Wang He Zhang Biao Pan Keni Qiu Wang Kang Jun Wang and Weisheng Zhao. 2024. Toward Energy-efficient STT-MRAM-based Near Memory Computing Architecture for Embedded Systems. ACM Transactions on Embedded Computing Systems 23 3 (2024) 1\u201324.","DOI":"10.1145\/3650729"},{"key":"e_1_3_3_1_22_2","unstructured":"Chien-Yu Lin Keisuke Kamahori Yiyu Liu Xiaoxiang Shi Madhav Kashyap Yile Gu Rulin Shao Zihao Ye Kan Zhu Stephanie Wang Arvind Krishnamurthy Rohan Kadekodi Luis Ceze and Baris Kasikci. 2025. TeleRAG: Efficient Retrieval-Augmented Generation Inference with Lookahead Retrieval. arxiv:https:\/\/arXiv.org\/abs\/2502.20969\u00a0[cs.IR]"},{"key":"e_1_3_3_1_23_2","doi-asserted-by":"crossref","unstructured":"Renze Lou Kai Zhang and Wenpeng Yin. 2024. Large language model instruction following: A survey of progresses and challenges. Computational Linguistics 50 3 (2024) 1053\u20131095.","DOI":"10.1162\/coli_a_00523"},{"key":"e_1_3_3_1_24_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.emnlp-main.334"},{"key":"e_1_3_3_1_25_2","doi-asserted-by":"crossref","unstructured":"Tommaso Marinelli Jos\u00e9 Ignacio\u00a0G\u00f3mez P\u00e9rez Christian Tenllado Manu Komalan Mohit Gupta and Francky Catthoor. 2022. Microarchitectural exploration of STT-MRAM last-level cache parameters for energy-efficient devices. ACM Transactions on Embedded Computing Systems (TECS) 21 1 (2022) 1\u201320.","DOI":"10.1145\/3490391"},{"key":"e_1_3_3_1_26_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.acl-main.173"},{"key":"e_1_3_3_1_27_2","unstructured":"NVIDIA Corporation. 2025. NVIDIA H100 Tensor Core GPU. https:\/\/www.nvidia.com\/en-us\/data-center\/h100\/. Accessed: 2025-11-18."},{"key":"e_1_3_3_1_28_2","doi-asserted-by":"crossref","unstructured":"Tae\u00a0Woo Oh Hanwool Jeong Kyoman Kang Juhyun Park Younghwi Yang and Seong-Ook Jung. 2016. Power-gated 9T SRAM cell for low-energy operation. IEEE Transactions on Very Large Scale Integration (VLSI) Systems 25 3 (2016) 1183\u20131187.","DOI":"10.1109\/TVLSI.2016.2623601"},{"key":"e_1_3_3_1_29_2","unstructured":"OpenAI. 2025. Introducing GPT-4.5. https:\/\/openai.com\/index\/introducing-gpt-4-5\/. Accessed: 2025-03-26."},{"key":"e_1_3_3_1_30_2","doi-asserted-by":"crossref","unstructured":"Nour Sayed Longfei Mao and Mehdi\u00a0B Tahoori. 2021. Dynamic behavior predictions for fast and efficient hybrid STT-MRAM caches. ACM Journal on Emerging Technologies in Computing Systems (JETC) 17 1 (2021) 1\u201321.","DOI":"10.1145\/3423135"},{"key":"e_1_3_3_1_31_2","doi-asserted-by":"crossref","unstructured":"Prabuddha Sinha Krishna\u00a0Prathik BV Shirshendu Das and Venkata\u00a0Kalyan Tavva. 2025. SmartDeCoup: Decoupling the STT-RAM LLC for even write distribution and lifetime improvement. Journal of Systems Architecture 159 (2025) 103237.","DOI":"10.1016\/j.sysarc.2025.103367"},{"key":"e_1_3_3_1_32_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.emnlp-main.566"},{"key":"e_1_3_3_1_33_2","doi-asserted-by":"publisher","DOI":"10.1109\/EPEPS61853.2024.10753867"},{"key":"e_1_3_3_1_34_2","doi-asserted-by":"publisher","DOI":"10.1145\/3676536.3697115"},{"key":"e_1_3_3_1_35_2","doi-asserted-by":"crossref","unstructured":"Wenchao Tian Bin Li Zhao Li Hao Cui Jing Shi Yongkun Wang and Jingrong Zhao. 2022. Using chiplet encapsulation technology to achieve processing-in-memory functions. Micromachines 13 10 (2022) 1790.","DOI":"10.3390\/mi13101790"},{"key":"e_1_3_3_1_36_2","unstructured":"UCIe Consortium. 2025. UCIe Specifications. https:\/\/www.uciexpress.org\/specifications. Accessed: 2025-09-13."},{"key":"e_1_3_3_1_37_2","unstructured":"Liang Wang Nan Yang Xiaolong Huang Binxing Jiao Linjun Yang Daxin Jiang Rangan Majumder and Furu Wei. 2022. Text Embeddings by Weakly-Supervised Contrastive Pre-training. arxiv:https:\/\/arXiv.org\/abs\/2212.03533\u00a0[cs.CL]"},{"key":"e_1_3_3_1_38_2","volume-title":"Proceedings of the International Conference on Learning Representations (ICLR)","author":"Wang Zilong","year":"2025","unstructured":"Zilong Wang, Zifeng Wang, Long\u00a0T. Le, Huaixiu\u00a0Steven Zheng, Swaroop Mishra, Vincent Perot, Yuwei Zhang, Anush Mattapalli, Ankur Taly, Jingbo Shang, Chen-Yu Lee, and Tomas Pfister. 2025. Speculative RAG: Enhancing Retrieval-Augmented Generation through Drafting. In Proceedings of the International Conference on Learning Representations (ICLR). OpenReview.net, Singapore."},{"key":"e_1_3_3_1_39_2","doi-asserted-by":"crossref","unstructured":"Cong Xu Yang Zheng Dimin Niu Xiaochun Zhu Seung\u00a0H Kang and Yuan Xie. 2015. Impact of write pulse and process variation on 22 nm FinFET-based STT-RAM design: A device-architecture co-optimization approach. IEEE Transactions on Multi-Scale Computing Systems 1 4 (2015) 195\u2013206.","DOI":"10.1109\/TMSCS.2015.2509960"},{"key":"e_1_3_3_1_40_2","doi-asserted-by":"crossref","unstructured":"Wang Xu and Israel Koren. 2024. A Scalable Wear Leveling Technique for Phase Change Memory. ACM Transactions on Storage 20 1 (2024) 1\u201326.","DOI":"10.1145\/3631146"},{"key":"e_1_3_3_1_41_2","doi-asserted-by":"publisher","DOI":"10.1109\/ISSCC42615.2023.10067752"},{"key":"e_1_3_3_1_42_2","doi-asserted-by":"publisher","DOI":"10.1109\/ISSCC42613.2021.9365945"},{"key":"e_1_3_3_1_43_2","doi-asserted-by":"crossref","unstructured":"Shimeng Yu and Tae-Hyeon Kim. 2024. Semiconductor memory technologies: State-of-the-art and future trends. Computer 57 4 (2024) 150\u2013154.","DOI":"10.1109\/MC.2024.3363269"},{"key":"e_1_3_3_1_44_2","doi-asserted-by":"crossref","unstructured":"Zitong Zhang Wenjie Wang Pingping Yu and Yanfeng Jiang. 2022. Cache performance of NV-STT-MRAM with scale effect and comparison with SRAM. International Journal of Electronics 109 3 (2022) 391\u2013409.","DOI":"10.1080\/00207217.2021.1908630"}],"event":{"name":"GLSVLSI '26: Great Lakes Symposium on VLSI 2026","location":"Canandaigua , NY , USA","acronym":"GLSVLSI '26","sponsor":["SIGDA ACM Special Interest Group on Design Automation","IEEE CEDA"]},"container-title":["Proceedings of the Great Lakes Symposium on VLSI 2026"],"original-title":[],"deposited":{"date-parts":[[2026,6,18]],"date-time":"2026-06-18T14:29:59Z","timestamp":1781792999000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3787109.3815268"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,6,22]]},"references-count":43,"alternative-id":["10.1145\/3787109.3815268","10.1145\/3787109"],"URL":"https:\/\/doi.org\/10.1145\/3787109.3815268","relation":{},"subject":[],"published":{"date-parts":[[2026,6,22]]},"assertion":[{"value":"2026-06-22","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}