{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,15]],"date-time":"2026-03-15T15:30:30Z","timestamp":1773588630103,"version":"3.50.1"},"publisher-location":"New York, NY, USA","reference-count":168,"publisher":"ACM","funder":[{"name":"National Research Foundation of Korea &#x28;NRF&#x29;","award":["2022R1C1C1011307"],"award-info":[{"award-number":["2022R1C1C1011307"]}]},{"name":"National Research Foundation of Korea &#x28;NRF&#x29;","award":["RS-2025-00519994"],"award-info":[{"award-number":["RS-2025-00519994"]}]},{"name":"Institute of Information communications Technology Planning Evaluation &#x28;IITP&#x29;","award":["RS-2024-00395134"],"award-info":[{"award-number":["RS-2024-00395134"]}]},{"name":"Institute of Information communications Technology Planning Evaluation &#x28;IITP&#x29;","award":["RS-2024-00347394"],"award-info":[{"award-number":["RS-2024-00347394"]}]},{"name":"Institute of Information communications Technology Planning Evaluation &#x28;IITP&#x29;","award":["RS-2023-00256081"],"award-info":[{"award-number":["RS-2023-00256081"]}]},{"name":"Institute of Information communications Technology Planning Evaluation &#x28;IITP&#x29;","award":["RS-2021-II211343"],"award-info":[{"award-number":["RS-2021-II211343"]}]},{"name":"Institute of Information communications Technology Planning Evaluation &#x28;IITP&#x29;","award":["RS-2024-00459026"],"award-info":[{"award-number":["RS-2024-00459026"]}]},{"name":"Samsung Memory Research Center &#x28;SMRC&#x29;","award":["None"],"award-info":[{"award-number":["None"]}]},{"name":"Korea Basic Science Institute &#x28;National research Facilities and Equipment Center&#x29;","award":["RS-2025-00564840"],"award-info":[{"award-number":["RS-2025-00564840"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,3,22]]},"DOI":"10.1145\/3779212.3790119","type":"proceedings-article","created":{"date-parts":[[2026,3,10]],"date-time":"2026-03-10T13:55:26Z","timestamp":1773150926000},"page":"5-24","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["A Cost-Effective Near-Storage Processing Solution for Offline Inference of Long-Context LLMs"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-4291-6124","authenticated-orcid":false,"given":"Hongsun","family":"Jang","sequence":"first","affiliation":[{"name":"Seoul National University, Seoul, Republic of Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9976-7487","authenticated-orcid":false,"given":"Jaeyong","family":"Song","sequence":"additional","affiliation":[{"name":"Seoul National University, Seoul, Republic of Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-8242-8330","authenticated-orcid":false,"given":"Changmin","family":"Shin","sequence":"additional","affiliation":[{"name":"Seoul National University, Seoul, Republic of Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-2228-2131","authenticated-orcid":false,"given":"Si Ung","family":"Noh","sequence":"additional","affiliation":[{"name":"Seoul National University, Seoul, Republic of Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0770-9277","authenticated-orcid":false,"given":"Jaewon","family":"Jung","sequence":"additional","affiliation":[{"name":"Seoul National University, Seoul, Republic of Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1826-9003","authenticated-orcid":false,"given":"Jisung","family":"Park","sequence":"additional","affiliation":[{"name":"POSTECH, Pohang, Republic of Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4010-6611","authenticated-orcid":false,"given":"Jinho","family":"Lee","sequence":"additional","affiliation":[{"name":"Seoul National University, Seoul, Republic of Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2026,3,22]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1145\/291069.291026"},{"key":"e_1_3_2_1_2_1","unstructured":"Josh Achiam Steven Adler Sandhini Agarwal Lama Ahmad Ilge Akkaya Florencia Leoni Aleman Diogo Almeida Janko Altenschmidt Sam Altman Shyamal Anadkat et al. 2023. GPT-4 Technical Report. arXiv preprint arXiv:2303.08774 (2023)."},{"key":"e_1_3_2_1_3_1","volume-title":"Sarathi: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills. arXiv preprint arXiv:2308.16369","author":"Agrawal Amey","year":"2023","unstructured":"Amey Agrawal, Ashish Panwar, Jayashree Mohan, Nipun Kwatra, Bhargav S Gulavani, and Ramachandran Ramjee. 2023. Sarathi: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills. arXiv preprint arXiv:2308.16369 (2023)."},{"key":"e_1_3_2_1_4_1","volume-title":"GQA: Training Generalized Multi-query Transformer Models from Multi-head Checkpoints. arXiv preprint arXiv:2305.13245","author":"Ainslie Joshua","year":"2023","unstructured":"Joshua Ainslie, James Lee-Thorp, Michiel de Jong, Yury Zemlyanskiy, Federico Lebr\u00f3n, and Sumit Sanghai. 2023. GQA: Training Generalized Multi-query Transformer Models from Multi-head Checkpoints. arXiv preprint arXiv:2305.13245 (2023)."},{"key":"e_1_3_2_1_5_1","volume-title":"Proceedings of Government Microcircuit Applications and Critical Technology Conference (GOMACTech).","author":"Ajayi T","year":"2019","unstructured":"T Ajayi, D Blaauw, TB Chan, CK Cheng, VA Chhabria, DK Choo, M Coltella, S Dobre, R Dreslinski, M Foga\u00e7a, et al., 2019a. OpenROAD: Toward a Self-Driving, Open-Source Digital Layout Implementation Tool Chain. In Proceedings of Government Microcircuit Applications and Critical Technology Conference (GOMACTech)."},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1145\/3316781.3326334"},{"key":"e_1_3_2_1_7_1","volume-title":"Mohammad Rastegari, and Mehrdad Farajtabar.","author":"Alizadeh Keivan","year":"2023","unstructured":"Keivan Alizadeh, Iman Mirzadeh, Dmitry Belenko, Karen Khatamifard, Minsik Cho, Carlo C Del Mundo, Mohammad Rastegari, and Mehrdad Farajtabar. 2023. Llm in a Flash: Efficient Large Language Model Inference with Limited Memory. arXiv preprint arXiv:2312.11514 (2023)."},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"crossref","unstructured":"Tyler Allen and Rong Ge. 2021. In-depth Analyses of Unified Virtual Memory System for GPU Accelerated Computing. In SC'21: Proceedings of the International Conference for High Performance Computing Networking Storage and Analysis (SC).","DOI":"10.1145\/3458817.3480855"},{"key":"e_1_3_2_1_9_1","unstructured":"AMD\/Samsung. [n.d.]. SmartSSD. https:\/\/www.xilinx.com\/publications\/product-briefs\/xilinx-smartssd-computational-storage-drive-product-brief.pdf. Accessed: 2025-12-12."},{"key":"e_1_3_2_1_10_1","unstructured":"AMD\/Xilinx. 2021. Vitis High-Level Synthesis User Guide (ug1399)."},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1109\/SC41404.2022.00051"},{"key":"e_1_3_2_1_12_1","unstructured":"ASPEED. [n.d.]. AST2500\/AST2520. https:\/\/vgamuseum.info\/images\/doc\/aspeed\/ast2520a2gp_datasheet.pdf. Accessed: 2025-12-12."},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1145\/2505515.2507847"},{"key":"e_1_3_2_1_14_1","volume-title":"Proceedings of the 19th USENIX Conference on File and Storage Technologies (FAST).","author":"Bae Jonghyun","unstructured":"Jonghyun Bae, Jongsung Lee, Yunho Jin, Sam Son, Shine Kim, Hakbeom Jang, Tae Jun Ham, and Jae W. Lee. 2021. FlashNeuron: SSD-Enabled Large-Batch Training of Very Deep Neural Networks. In Proceedings of the 19th USENIX Conference on File and Storage Technologies (FAST)."},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-long.172"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1145\/3085572"},{"key":"e_1_3_2_1_17_1","article-title":"SPIN: Seamless Operating System Integration of Peer-to-Peer DMA between SSDs and GPUs","volume":"36","author":"Bergman Shai","year":"2019","unstructured":"Shai Bergman, Tanya Brokhman, Tzachi Cohen, and Mark Silberstein. 2019. SPIN: Seamless Operating System Integration of Peer-to-Peer DMA between SSDs and GPUs. ACM Trans. Comput. Syst., Vol. 36, 2, Article 5 (2019), 26 pages.","journal-title":"ACM Trans. Comput. Syst."},{"key":"e_1_3_2_1_18_1","volume-title":"Petals: Collaborative Inference and Fine-Tuning of Large Models. arXiv preprint arXiv:2209.01188","author":"Borzunov Alexander","year":"2022","unstructured":"Alexander Borzunov, Dmitry Baranchuk, Tim Dettmers, Max Ryabinin, Younes Belkada, Artem Chumachenko, Pavel Samygin, and Colin Raffel. 2022. Petals: Collaborative Inference and Fine-Tuning of Large Models. arXiv preprint arXiv:2209.01188 (2022)."},{"key":"e_1_3_2_1_19_1","volume-title":"Proceedings of the Advances in Neural Information Processing Systems 33 (NeurIPS).","author":"Brown Tom","year":"2020","unstructured":"Tom Brown, Benjamin Mann, Nick Ryder, Melanie Subbiah, Jared D Kaplan, Prafulla Dhariwal, Arvind Neelakantan, Pranav Shyam, Girish Sastry, Amanda Askell, Sandhini Agarwal, Ariel Herbert-Voss, Gretchen Krueger, Tom Henighan, Rewon Child, Aditya Ramesh, Daniel Ziegler, Jeffrey Wu, Clemens Winter, Chris Hesse, Mark Chen, Eric Sigler, Mateusz Litwin, Scott Gray, Benjamin Chess, Jack Clark, Christopher Berner, Sam McCandlish, Alec Radford, Ilya Sutskever, and Dario Amodei. 2020. Language Models are Few-Shot Learners. In Proceedings of the Advances in Neural Information Processing Systems 33 (NeurIPS)."},{"key":"e_1_3_2_1_20_1","volume-title":"Proceedings of the 30th International Conference on Architectural Support for Programming Languages and Operating Systems (ASPLOS).","author":"Cao Shiyi","year":"2025","unstructured":"Shiyi Cao, Shu Liu, Tyler Griggs, Peter Schafhalter, Xiaoxuan Liu, Ying Sheng, Joseph E. Gonzalez, Matei Zaharia, and Ion Stoica. 2025. MoE-Lightning: High-Throughput MoE Inference on Memory-Constrained GPUs. In Proceedings of the 30th International Conference on Architectural Support for Programming Languages and Operating Systems (ASPLOS)."},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1145\/3620666.3651353"},{"key":"e_1_3_2_1_22_1","volume-title":"Proceedings of the 12th International Conference on Learning Representations (ICLR).","author":"Chang Yapei","year":"2024","unstructured":"Yapei Chang, Kyle Lo, Tanya Goyal, and Mohit Iyyer. 2024b. BooookScore: A Systematic Exploration of Book-Length Summarization in the Era of LLMs. In Proceedings of the 12th International Conference on Learning Representations (ICLR)."},{"key":"e_1_3_2_1_23_1","article-title":"Understanding the Potential of FPGA-based Spatial Acceleration for Large Language Model Inference","volume":"18","author":"Chen Hongzheng","year":"2024","unstructured":"Hongzheng Chen, Jiahao Zhang, Yixiao Du, Shaojie Xiang, Zichao Yue, Niansong Zhang, Yaohui Cai, and Zhiru Zhang. 2024b. Understanding the Potential of FPGA-based Spatial Acceleration for Large Language Model Inference. ACM Trans. Reconfigurable Technol. Syst., Vol. 18, 1, Article 5 (2024), 29 pages.","journal-title":"ACM Trans. Reconfigurable Technol. Syst."},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1145\/3695053.3731116"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1145\/3627703.3629553"},{"key":"e_1_3_2_1_26_1","unstructured":"CRZ Technology Inc. [n.d.]. Daisplus Openssd. https:\/\/www.crz-tech.com\/crz\/article\/DaisyPlus\/. Accessed: 2025-12-12."},{"key":"e_1_3_2_1_27_1","volume-title":"Proceedings of the 2022 USENIX Annual Technical Conference (USENIX ATC).","author":"Cui Weihao","year":"2022","unstructured":"Weihao Cui, Han Zhao, Quan Chen, Hao Wei, Zirui Li, Deze Zeng, Chao Li, and Minyi Guo. 2022. DVABatch: Diversity-aware Multi-Entry Multi-Exit Batching for Efficient Processing of DNN Services on GPUs. In Proceedings of the 2022 USENIX Annual Technical Conference (USENIX ATC)."},{"key":"e_1_3_2_1_28_1","unstructured":"CXL Consortium. [n.d.]. Compute Express Link. https:\/\/www.computeexpresslink.org. Accessed: 2025-12-12."},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P19-1285"},{"key":"e_1_3_2_1_30_1","volume-title":"Proceedings of the 12th International Conference on Learning Representations (ICLR).","author":"Dao Tri","year":"2024","unstructured":"Tri Dao. 2024. FlashAttention-2: Faster Attention with Better Parallelism and Work Partitioning. In Proceedings of the 12th International Conference on Learning Representations (ICLR)."},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.52202\/068431-1189"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1145\/3669900"},{"key":"e_1_3_2_1_33_1","volume-title":"Proceedings of the Advances in Neural Information Processing Systems 35 (NeurIPS).","author":"Dettmers Tim","year":"2022","unstructured":"Tim Dettmers, Mike Lewis, Younes Belkada, and Luke Zettlemoyer. 2022. GPT3.int8(): 8-bit Matrix Multiplication for Transformers at Scale. In Proceedings of the Advances in Neural Information Processing Systems 35 (NeurIPS)."},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/HOTCHIPS.2019.8875680"},{"key":"e_1_3_2_1_35_1","volume-title":"Proceedings of the 2013 SIGMOD International Conference on Management of Data (SIGMOD).","author":"Do Jaeyoung","unstructured":"Jaeyoung Do, Yang-Suk Kee, Jignesh M. Patel, Chanik Park, Kwanghyun Park, and David J. DeWitt. 2013. Query Processing on Smart SSDs: Opportunities and Challenges. In Proceedings of the 2013 SIGMOD International Conference on Management of Data (SIGMOD)."},{"key":"e_1_3_2_1_36_1","volume-title":"Proceedings of the 39th International Conference on Machine Learning (ICML).","author":"Du Nan","year":"2022","unstructured":"Nan Du, Yanping Huang, Andrew M Dai, Simon Tong, Dmitry Lepikhin, Yuanzhong Xu, Maxim Krikun, Yanqi Zhou, Adams Wei Yu, Orhan Firat, Barret Zoph, Liam Fedus, Maarten P Bosma, Zongwei Zhou, Tao Wang, Emma Wang, Kellie Webster, Marie Pellat, Kevin Robinson, Kathleen Meier-Hellstern, Toju Duke, Lucas Dixon, Kun Zhang, Quoc Le, Yonghui Wu, Zhifeng Chen, and Claire Cui. 2022. GLaM: Efficient Scaling of Language Models with Mixture-of-Experts. In Proceedings of the 39th International Conference on Machine Learning (ICML)."},{"key":"e_1_3_2_1_37_1","unstructured":"Abhimanyu Dubey Abhinav Jauhri Abhinav Pandey Abhishek Kadian Ahmad Al-Dahle Aiesha Letman Akhil Mathur Alan Schelten Amy Yang Angela Fan et al. 2024. The Llama 3 Herd of Models. arXiv preprint arXiv:2407.21783 (2024)."},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1109\/MCSE.2022.3163817"},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.5555\/3691992.3691999"},{"key":"e_1_3_2_1_40_1","unstructured":"Leo Gao Jonathan Tow Stella Biderman Sid Black Anthony DiPofi Charles Foster Laurence Golding Jeffrey Hsu Kyle McDonell Niklas Muennighoff et al. 2021. A Framework for Few-Shot Language Model Evaluation. Zenodo (2021)."},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA59077.2024.00054"},{"key":"e_1_3_2_1_42_1","unstructured":"Github. [n.d.]. Github Copilot. https:\/\/github.com\/features\/copilot\/. Accessed: 2025-12-12."},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1109\/MM.2023.3237491"},{"key":"e_1_3_2_1_44_1","volume-title":"Junyi Jessy Li, and Greg Durrett","author":"Goyal Tanya","year":"2022","unstructured":"Tanya Goyal, Junyi Jessy Li, and Greg Durrett. 2022. News Summarization and Evaluation in the Era of Gpt-3. arXiv preprint arXiv:2209.12356 (2022)."},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA.2016.23"},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.2139\/ssrn.4583531"},{"key":"e_1_3_2_1_47_1","volume-title":"Gpt4graph: Can Large Language Models Understand Graph Structured Data? An Empirical Evaluation and Benchmarking. arXiv preprint arXiv:2305.15066","author":"Guo Jiayan","year":"2023","unstructured":"Jiayan Guo, Lun Du, and Hengyu Liu. 2023. Gpt4graph: Can Large Language Models Understand Graph Structured Data? An Empirical Evaluation and Benchmarking. arXiv preprint arXiv:2305.15066 (2023)."},{"key":"e_1_3_2_1_48_1","unstructured":"H3. [n.d.]. Falcon 4109. https:\/\/www.h3platform.com\/product-detail\/overview\/25. Accessed: 2025-12-12."},{"key":"e_1_3_2_1_49_1","volume-title":"Proceedings of the 53th Annual International Symposium on Microarchitecture (MICRO).","author":"He Mingxuan","unstructured":"Mingxuan He, Choungki Song, Ilkon Kim, Chunseok Jeong, Seho Kim, Il Park, Mithuna Thottethodi, and T. N. Vijaykumar. 2020. Newton: A DRAM-maker's Accelerator-in-Memory (AiM) Architecture for Machine Learning. In Proceedings of the 53th Annual International Symposium on Microarchitecture (MICRO)."},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"publisher","DOI":"10.1145\/3620666.3651380"},{"key":"e_1_3_2_1_51_1","volume-title":"Jeff Rasley, Samyam Rajbhandari, Reza Yazdani Aminabadi, Heyang Qin, Arash Bakhtiari, Lev Kurilenko, et al.","author":"Holmes Connor","year":"2024","unstructured":"Connor Holmes, Masahiro Tanaka, Michael Wyatt, Ammar Ahmad Awan, Jeff Rasley, Samyam Rajbhandari, Reza Yazdani Aminabadi, Heyang Qin, Arash Bakhtiari, Lev Kurilenko, et al., 2024. Deepspeed-Fastgen: High-Throughput Text Generation for LLMs via MII and Deepspeed-Inference. arXiv preprint arXiv:2401.08671 (2024)."},{"key":"e_1_3_2_1_52_1","doi-asserted-by":"publisher","DOI":"10.1109\/MICRO56248.2022.00051"},{"key":"e_1_3_2_1_53_1","volume-title":"Kurt Keutzer, and Amir Gholami.","author":"Hooper Coleman","year":"2024","unstructured":"Coleman Hooper, Sehoon Kim, Hiva Mohammadzadeh, Michael W Mahoney, Yakun Sophia Shao, Kurt Keutzer, and Amir Gholami. 2024. KVQuant: Towards 10 Million Context Length LLM Inference with KV Cache Quantization. arXiv preprint arXiv:2401.18079 (2024)."},{"key":"e_1_3_2_1_54_1","volume-title":"Proceedings of the Advances in Neural Information Processing Systems 32 (NeurIPS).","author":"Huang Yanping","year":"2019","unstructured":"Yanping Huang, Youlong Cheng, Ankur Bapna, Orhan Firat, Dehao Chen, Mia Chen, HyoukJoong Lee, Jiquan Ngiam, Quoc V Le, Yonghui Wu, and zhifeng Chen. 2019. GPipe: Efficient Training of Giant Neural Networks using Pipeline Parallelism. In Proceedings of the Advances in Neural Information Processing Systems 32 (NeurIPS)."},{"key":"e_1_3_2_1_55_1","unstructured":"Hugging Face. [n.d.]. Hugging Face. https:\/\/huggingface.co\/docs\/hub\/index. Accessed: 2025-12-12."},{"key":"e_1_3_2_1_56_1","doi-asserted-by":"publisher","DOI":"10.23919\/DATE56975.2023.10137044"},{"key":"e_1_3_2_1_57_1","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA57654.2024.00034"},{"key":"e_1_3_2_1_58_1","volume-title":"Proceedings of the 2023 USENIX Annual Technical Conference (USENIX ATC).","author":"Jang Junhyeok","year":"2023","unstructured":"Junhyeok Jang, Hanjin Choi, Hanyeoreum Bae, Seungjun Lee, Miryeong Kwon, and Myoungsoo Jung. 2023a. CXL-ANNS: Software-Hardware Collaborative Memory Disaggregation and Computation for Billion-Scale Approximate Nearest Neighbor Search. In Proceedings of the 2023 USENIX Annual Technical Conference (USENIX ATC)."},{"key":"e_1_3_2_1_59_1","doi-asserted-by":"publisher","DOI":"10.1145\/3676641.3716245"},{"key":"e_1_3_2_1_60_1","volume-title":"Diego de las Casas, Emma Bou Hanna, Florian Bressand, et al.","author":"Jiang Albert Q","year":"2024","unstructured":"Albert Q Jiang, Alexandre Sablayrolles, Antoine Roux, Arthur Mesch, Blanche Savary, Chris Bamford, Devendra Singh Chaplot, Diego de las Casas, Emma Bou Hanna, Florian Bressand, et al., 2024. Mixtral of Experts. arXiv preprint arXiv:2401.04088 (2024)."},{"key":"e_1_3_2_1_61_1","volume-title":"Tae Jun Ham, and Jae W. Lee","author":"Jin Yunho","year":"2022","unstructured":"Yunho Jin, Shine Kim, Tae Jun Ham, and Jae W. Lee. 2022. Architecting a Flash-Based Storage System for Low-Cost Inference of Extreme-Scale DNNs. IEEE Trans. Comput. (2022)."},{"key":"e_1_3_2_1_62_1","doi-asserted-by":"publisher","DOI":"10.1109\/MSST.2013.6558444"},{"key":"e_1_3_2_1_63_1","doi-asserted-by":"publisher","DOI":"10.1145\/290593.290602"},{"key":"e_1_3_2_1_64_1","doi-asserted-by":"publisher","DOI":"10.1145\/3431920.3439477"},{"key":"e_1_3_2_1_65_1","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA51647.2021.00030"},{"key":"e_1_3_2_1_66_1","doi-asserted-by":"publisher","DOI":"10.1145\/3695053.3731092"},{"key":"e_1_3_2_1_67_1","doi-asserted-by":"publisher","DOI":"10.1145\/3061639.3062264"},{"key":"e_1_3_2_1_68_1","volume-title":"Proceedings of the 21st USENIX Conference on File and Storage Technologies (FAST).","author":"Kim Sang-Hoon","year":"2023","unstructured":"Sang-Hoon Kim, Jaehoon Shim, Euidong Lee, Seongyeop Jeong, Ilkueon Kang, and Jin-Soo Kim. 2023. NVMeVirt: A Versatile Software-defined Virtual NVMe Device. In Proceedings of the 21st USENIX Conference on File and Storage Technologies (FAST)."},{"key":"e_1_3_2_1_69_1","volume-title":"Proceedings of the 2024 International Conference on Networking and Communications (ICNWC).","author":"Kumar P.","unstructured":"P. Kumar, S. Manikandan, and R. Kishore. 2024. AI-Driven Text Generation: A Novel GPT-Based Approach for Automated Content Creation. In Proceedings of the 2024 International Conference on Networking and Communications (ICNWC)."},{"key":"e_1_3_2_1_70_1","article-title":"Cosmos OpenSSD","volume":"16","author":"Kwak Jaewook","year":"2020","unstructured":"Jaewook Kwak, Sangjin Lee, Kibin Park, Jinwoo Jeong, and Yong Ho Song. 2020. Cosmos OpenSSD: Rapid Prototype for Flash Storage Systems. ACM Trans. Storage, Vol. 16, 3, Article 15 (2020), 35 pages.","journal-title":"Rapid Prototype for Flash Storage Systems. ACM Trans. Storage"},{"key":"e_1_3_2_1_71_1","doi-asserted-by":"publisher","DOI":"10.1145\/3600006.3613165"},{"key":"e_1_3_2_1_72_1","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA61900.2025.00126"},{"key":"e_1_3_2_1_73_1","doi-asserted-by":"publisher","DOI":"10.1145\/3695053.3731073"},{"key":"e_1_3_2_1_74_1","doi-asserted-by":"publisher","DOI":"10.14778\/3137765.3137776"},{"key":"e_1_3_2_1_75_1","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA52012.2021.00013"},{"key":"e_1_3_2_1_76_1","doi-asserted-by":"publisher","DOI":"10.1109\/ISSCC42614.2022.9731711"},{"key":"e_1_3_2_1_77_1","volume-title":"Proceedings of the 18th USENIX Symposium on Operating Systems Design and Implementation (OSDI).","author":"Lee Wonbeom","year":"2024","unstructured":"Wonbeom Lee, Jungi Lee, Junghwan Seo, and Jaewoong Sim. 2024. InfiniGen: Efficient Generative Inference of Large Language Models with Dynamic KV Cache Management. In Proceedings of the 18th USENIX Symposium on Operating Systems Design and Implementation (OSDI)."},{"key":"e_1_3_2_1_78_1","doi-asserted-by":"publisher","DOI":"10.1145\/3470496.3527391"},{"key":"e_1_3_2_1_79_1","doi-asserted-by":"publisher","DOI":"10.1145\/3370748.3406567"},{"key":"e_1_3_2_1_80_1","doi-asserted-by":"publisher","DOI":"10.1145\/3620665.3640376"},{"key":"e_1_3_2_1_81_1","volume-title":"Sang Michael Xie, Shibani Santurkar, Surya Ganguli, Tatsunori Hashimoto, Thomas Icard, Tianyi Zhang, Vishrav Chaudhary, William Wang, Xuechen Li, Yifan Mai, Yuhui Zhang, and Yuta Koreeda.","author":"Liang Percy","year":"2023","unstructured":"Percy Liang, Rishi Bommasani, Tony Lee, Dimitris Tsipras, Dilara Soylu, Michihiro Yasunaga, Yian Zhang, Deepak Narayanan, Yuhuai Wu, Ananya Kumar, Benjamin Newman, Binhang Yuan, Bobby Yan, Ce Zhang, Christian Cosgrove, Christopher D Manning, Christopher Re, Diana Acosta-Navas, Drew A. Hudson, Eric Zelikman, Esin Durmus, Faisal Ladhak, Frieda Rong, Hongyu Ren, Huaxiu Yao, Jue WANG, Keshav Santhanam, Laurel Orr, Lucia Zheng, Mert Yuksekgonul, Mirac Suzgun, Nathan Kim, Neel Guha, Niladri S. Chatterji, Omar Khattab, Peter Henderson, Qian Huang, Ryan Andrew Chi, Sang Michael Xie, Shibani Santurkar, Surya Ganguli, Tatsunori Hashimoto, Thomas Icard, Tianyi Zhang, Vishrav Chaudhary, William Wang, Xuechen Li, Yifan Mai, Yuhui Zhang, and Yuta Koreeda. 2023. Holistic Evaluation of Language Models. Transactions on Machine Learning Research (2023)."},{"key":"e_1_3_2_1_82_1","doi-asserted-by":"publisher","DOI":"10.1145\/3470496.3527433"},{"key":"e_1_3_2_1_83_1","unstructured":"Aixin Liu Bei Feng Bing Xue Bingxuan Wang Bochao Wu Chengda Lu Chenggang Zhao Chengqi Deng Chenyu Zhang Chong Ruan et al. 2024a. Deepseek-v3 technical report. arXiv preprint arXiv:2412.19437 (2024)."},{"key":"e_1_3_2_1_84_1","volume-title":"Proceedings of the Advances in Neural Information Processing Systems 36 (NeurIPS).","author":"Liu Hao","year":"2023","unstructured":"Hao Liu and Pieter Abbeel. 2023. Blockwise Parallel Transformers for Large Context Models. In Proceedings of the Advances in Neural Information Processing Systems 36 (NeurIPS)."},{"key":"e_1_3_2_1_85_1","volume-title":"Proceedings of the 12th International Conference on Learning Representations (ICLR).","author":"Liu Hao","year":"2024","unstructured":"Hao Liu, Matei Zaharia, and Pieter Abbeel. 2024b. RingAttention with Blockwise Transformers for Near-Infinite Context. In Proceedings of the 12th International Conference on Learning Representations (ICLR)."},{"key":"e_1_3_2_1_86_1","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA61900.2025.00129"},{"key":"e_1_3_2_1_87_1","volume-title":"LLM-QAT: Data-free Quantization Aware Training for Large Language Models. arXiv preprint arXiv:2305.17888","author":"Liu Zechun","year":"2023","unstructured":"Zechun Liu, Barlas Oguz, Changsheng Zhao, Ernie Chang, Pierre Stock, Yashar Mehdad, Yangyang Shi, Raghuraman Krishnamoorthi, and Vikas Chandra. 2023. LLM-QAT: Data-free Quantization Aware Training for Large Language Models. arXiv preprint arXiv:2305.17888 (2023)."},{"key":"e_1_3_2_1_88_1","doi-asserted-by":"publisher","DOI":"10.1145\/3352460.3358320"},{"key":"e_1_3_2_1_89_1","volume-title":"Proceedings of the 27th International Conference on Architectural Support for Programming Languages and Operating Systems (ASPLOS).","author":"Ghiasi Nika Mansouri","year":"2022","unstructured":"Nika Mansouri Ghiasi, Jisung Park, Harun Mustafa, Jeremie Kim, Ataberk Olgun, Arvid Gollwitzer, Damla Senol Cali, Can Firtina, Haiyu Mao, Nour Almadhoun Alserr, Rachata Ausavarungnirun, Nandita Vijaykumar, Mohammed Alser, and Onur Mutlu. 2022. GenStore: A High-Performance In-Storage Processing System for Genome Sequence Analysis. In Proceedings of the 27th International Conference on Architectural Support for Programming Languages and Operating Systems (ASPLOS)."},{"key":"e_1_3_2_1_90_1","unstructured":"Carver Mead and Lynn Conway. 1980. Introduction to VLSI systems."},{"key":"e_1_3_2_1_91_1","volume-title":"Proceedings of the 5th International Conference on Learning Representations (ICLR).","author":"Merity Stephen","year":"2017","unstructured":"Stephen Merity, Caiming Xiong, James Bradbury, and Richard Socher. 2017. Pointer Sentinel Mixture Models. In Proceedings of the 5th International Conference on Learning Representations (ICLR)."},{"key":"e_1_3_2_1_92_1","unstructured":"Meta. [n.d.]. The Llama 4 herd: The beginning of a new era of natively multimodal AI innovation. Accessed: 2025-12-12."},{"key":"e_1_3_2_1_93_1","unstructured":"Micron. [n.d.]. Crucial T500 1 TB. https:\/\/www.techpowerup.com\/ssd-specs\/crucial-t500-1-tb.d1771. Accessed: 2025-12-12."},{"key":"e_1_3_2_1_94_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D18-1260"},{"key":"e_1_3_2_1_95_1","volume-title":"Online Normalizer Calculation for Softmax. arXiv preprint arXiv:1805.02867","author":"Milakov Maxim","year":"2018","unstructured":"Maxim Milakov and Natalia Gimelshein. 2018. Online Normalizer Calculation for Softmax. arXiv preprint arXiv:1805.02867 (2018)."},{"key":"e_1_3_2_1_96_1","volume-title":"Just the Summary! Topic-Aware Convolutional Neural Networks for Extreme Summarization. arXiv preprint arXiv:1808.08745","author":"Narayan Shashi","year":"2018","unstructured":"Shashi Narayan, Shay B Cohen, and Mirella Lapata. 2018. Don't Give Me the Details, Just the Summary! Topic-Aware Convolutional Neural Networks for Extreme Summarization. arXiv preprint arXiv:1808.08745 (2018)."},{"key":"e_1_3_2_1_97_1","unstructured":"Neil Brown. [n.d.]. mdadm. https:\/\/git.kernel.org\/pub\/scm\/utils\/mdadm\/mdadm.git\/. Accessed: 2025-12-12."},{"key":"e_1_3_2_1_98_1","unstructured":"NVIDIA. [n.d.]. GPUDirect Storage. https:\/\/developer.nvidia.com\/gpudirect-storage. Accessed: 2025-12-12."},{"key":"e_1_3_2_1_99_1","unstructured":"NVIDIA. [n.d.]. NVIDIA Management Library (NVML). https:\/\/developer.nvidia.com\/management-library-nvml. Accessed: 2025-12-12."},{"key":"e_1_3_2_1_100_1","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA61900.2025.00113"},{"key":"e_1_3_2_1_101_1","doi-asserted-by":"crossref","unstructured":"Bo Pang Erik Nijkamp Wojciech Kryscinski Silvio Savarese Yingbo Zhou and Caiming Xiong. 2023. Long Document Summarization with Top-down and Bottom-up Inference. In Findings of the Association for Computational Linguistics (ACL Findings).","DOI":"10.18653\/v1\/2023.findings-eacl.94"},{"key":"e_1_3_2_1_102_1","doi-asserted-by":"publisher","DOI":"10.1145\/3620665.3640422"},{"key":"e_1_3_2_1_103_1","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA57654.2024.00078"},{"key":"e_1_3_2_1_104_1","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA59077.2024.00019"},{"key":"e_1_3_2_1_105_1","unstructured":"PCI-SIG. [n.d.]. PCI express base specification 4.0. https:\/\/pcisig.com\/specifications. Accessed: 2025-12-12."},{"key":"e_1_3_2_1_106_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICFPT51103.2020.00016"},{"key":"e_1_3_2_1_107_1","volume-title":"Proceedings of the Machine Learning and Systems 5 (MLSys).","author":"Pope Reiner","year":"2023","unstructured":"Reiner Pope, Sholto Douglas, Aakanksha Chowdhery, Jacob Devlin, James Bradbury, Jonathan Heek, Kefan Xiao, Shivani Agrawal, and Jeff Dean. 2023. Efficiently scaling transformer inference. In Proceedings of the Machine Learning and Systems 5 (MLSys)."},{"key":"e_1_3_2_1_108_1","volume-title":"Summarization is (Almost) Dead. arXiv preprint arXiv:2309.09558","author":"Pu Xiao","year":"2023","unstructured":"Xiao Pu, Mingqi Gao, and Xiaojun Wan. 2023. Summarization is (Almost) Dead. arXiv preprint arXiv:2309.09558 (2023)."},{"key":"e_1_3_2_1_109_1","unstructured":"pybind. [n.d.]. pybind11. https:\/\/github.com\/pybind\/pybind11. Accessed: 2025-12-12."},{"key":"e_1_3_2_1_110_1","unstructured":"Python Software Foundation. [n.d.]. Distutils. https:\/\/docs.python.org\/3.9\/library\/distutils.html. Accessed: 2025-12-12."},{"key":"e_1_3_2_1_111_1","doi-asserted-by":"publisher","DOI":"10.1109\/MSST.2015.7208281"},{"key":"e_1_3_2_1_112_1","doi-asserted-by":"publisher","DOI":"10.1145\/3575693.3575748"},{"key":"e_1_3_2_1_113_1","volume-title":"Language Models Are Unsupervised Multitask Learners. OpenAI blog","author":"Radford Alec","year":"2019","unstructured":"Alec Radford, Jeffrey Wu, Rewon Child, David Luan, Dario Amodei, and Ilya Sutskever. 2019. Language Models Are Unsupervised Multitask Learners. OpenAI blog (2019)."},{"key":"e_1_3_2_1_114_1","doi-asserted-by":"publisher","DOI":"10.1145\/3458817.3476205"},{"key":"e_1_3_2_1_115_1","volume-title":"Proceedings of the 2021 USENIX Annual Technical Conference (USENIX ATC).","author":"Ren Jie","year":"2021","unstructured":"Jie Ren, Samyam Rajbhandari, Reza Yazdani Aminabadi, Olatunji Ruwase, Shuangyan Yang, Minjia Zhang, Dong Li, and Yuxiong He. 2021. Zero-offload: democratizing billion-scale model training. In Proceedings of the 2021 USENIX Annual Technical Conference (USENIX ATC)."},{"key":"e_1_3_2_1_116_1","volume-title":"Proceedings of the 41st International Conference on Machine Learning (ICML).","author":"Ribar Luka","year":"2024","unstructured":"Luka Ribar, Ivan Chelombiev, Luke Hudlass-Galley, Charlie Blake, Carlo Luschi, and Douglas Orr. 2024. SparQ Attention: Bandwidth-Efficient LLM Inference. In Proceedings of the 41st International Conference on Machine Learning (ICML)."},{"key":"e_1_3_2_1_117_1","doi-asserted-by":"publisher","DOI":"10.1109\/2.928624"},{"key":"e_1_3_2_1_118_1","doi-asserted-by":"publisher","DOI":"10.5555\/645924.671345"},{"key":"e_1_3_2_1_119_1","volume-title":"Cosmin Adrian Bejan, and Andrew S Gordon","author":"Roemmele Melissa","year":"2011","unstructured":"Melissa Roemmele, Cosmin Adrian Bejan, and Andrew S Gordon. 2011. Choice of Plausible Alternatives: An Evaluation of Commonsense Causal Reasoning.. In AAAI spring symposium: logical formalizations of commonsense reasoning."},{"key":"e_1_3_2_1_120_1","doi-asserted-by":"publisher","DOI":"10.1109\/JSSC.2015.2458972"},{"key":"e_1_3_2_1_121_1","unstructured":"Samsung. [n.d.]. PM9A3 3.8TB. https:\/\/www.techpowerup.com\/ssd-specs\/samsung-pm9a3-3-8-tb.d1255. Accessed: 2025-12-12."},{"key":"e_1_3_2_1_122_1","unstructured":"Samsung. [n.d.]. Samsung 128 GB DDR4 3200 LRDIMM ECC. https:\/\/semiconductor.samsung.com\/dram\/module\/lrdimm\/m386aag40am3-cwe\/ . Accessed: 2025-12-12."},{"key":"e_1_3_2_1_123_1","unstructured":"Samsung. [n.d.]. Samsung 990 Pro 1 TB. https:\/\/www.techpowerup.com\/ssd-specs\/samsung-990-pro-w-heatsink-1-tb.d1899. Accessed: 2025-12-12."},{"key":"e_1_3_2_1_124_1","doi-asserted-by":"publisher","DOI":"10.1145\/3342195.3387557"},{"key":"e_1_3_2_1_125_1","doi-asserted-by":"publisher","DOI":"10.1145\/3552326.3587456"},{"key":"e_1_3_2_1_126_1","volume-title":"Proceedings of the 2025 International Symposium on High Performance Computer Architecture (HPCA).","author":"Seo Seong Hoon","unstructured":"Seong Hoon Seo, Junghoon Kim, Donghyun Lee, Seonah Yoo, Seokwon Moon, Yeonhong Park, and Jae W. Lee. 2025. FACIL: Flexible DRAM Address Mapping for SoC-PIM Cooperative On-device LLM Inference. In Proceedings of the 2025 International Symposium on High Performance Computer Architecture (HPCA)."},{"key":"e_1_3_2_1_127_1","volume-title":"Proceedings of the 40th International Conference on Machine Learning (ICML).","author":"Sheng Ying","year":"2023","unstructured":"Ying Sheng, Lianmin Zheng, Binhang Yuan, Zhuohan Li, Max Ryabinin, Beidi Chen, Percy Liang, Christopher Re, Ion Stoica, and Ce Zhang. 2023. FlexGen: High-Throughput Generative Inference of Large Language Models with a Single GPU. In Proceedings of the 40th International Conference on Machine Learning (ICML)."},{"key":"e_1_3_2_1_128_1","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA61900.2025.00055"},{"key":"e_1_3_2_1_129_1","doi-asserted-by":"publisher","DOI":"10.1145\/3725843.3756127"},{"key":"e_1_3_2_1_130_1","volume-title":"Megatron-LM: Training Multi-Billion Parameter Language Models using Model Parallelism. arXiv preprint arXiv:1909.08053","author":"Shoeybi Mohammad","year":"2019","unstructured":"Mohammad Shoeybi, Mostofa Patwary, Raul Puri, Patrick LeGresley, Jared Casper, and Bryan Catanzaro. 2019. Megatron-LM: Training Multi-Billion Parameter Language Models using Model Parallelism. arXiv preprint arXiv:1909.08053 (2019)."},{"key":"e_1_3_2_1_131_1","doi-asserted-by":"publisher","DOI":"10.5555\/2093889.2093965"},{"key":"e_1_3_2_1_132_1","doi-asserted-by":"publisher","DOI":"10.1145\/3489525.3511672"},{"key":"e_1_3_2_1_133_1","volume-title":"Powerinfer: Fast large language model serving with a consumer-grade gpu. arXiv preprint arXiv:2312.12456","author":"Song Yixin","year":"2023","unstructured":"Yixin Song, Zeyu Mi, Haotong Xie, and Haibo Chen. 2023. Powerinfer: Fast large language model serving with a consumer-grade gpu. arXiv preprint arXiv:2312.12456 (2023)."},{"key":"e_1_3_2_1_134_1","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA61900.2025.00102"},{"key":"e_1_3_2_1_135_1","first-page":"127063","article-title":"Roformer","volume":"568","author":"Su Jianlin","year":"2024","unstructured":"Jianlin Su, Murtadha Ahmed, Yu Lu, Shengfeng Pan, Wen Bo, and Yunfeng Liu. 2024. Roformer: Enhanced Transformer with Rotary Position Embedding. Neurocomputing, Vol. 568 (2024), 127063.","journal-title":"Enhanced Transformer with Rotary Position Embedding. Neurocomputing"},{"key":"e_1_3_2_1_136_1","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA61900.2025.00128"},{"key":"e_1_3_2_1_137_1","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA53966.2022.00081"},{"key":"e_1_3_2_1_138_1","volume-title":"Proceedings of the 41st International Conference on Machine Learning (ICML).","author":"Tang Jiaming","year":"2024","unstructured":"Jiaming Tang, Yilong Zhao, Kan Zhu, Guangxuan Xiao, Baris Kasikci, and Song Han. 2024. QUEST: Query-Aware Sparsity for Efficient Long-Context LLM Inference. In Proceedings of the 41st International Conference on Machine Learning (ICML)."},{"key":"e_1_3_2_1_139_1","volume-title":"Ryan Burnell, Libin Bai, Anmol Gulati, Garrett Tanzer, Damien Vincent, Zhufeng Pan, Shibo Wang, et al.","author":"Team Gemini","year":"2024","unstructured":"Gemini Team, Petko Georgiev, Ving Ian Lei, Ryan Burnell, Libin Bai, Anmol Gulati, Garrett Tanzer, Damien Vincent, Zhufeng Pan, Shibo Wang, et al., 2024. Gemini 1.5: Unlocking Multimodal Understanding Across Millions of Tokens of Context. arXiv preprint arXiv:2403.05530 (2024)."},{"key":"e_1_3_2_1_140_1","unstructured":"Qwen Team. 2024. Qwen2 Technical Report. arXiv preprint arXiv:2407.10671 (2024)."},{"key":"e_1_3_2_1_141_1","doi-asserted-by":"publisher","DOI":"10.5555\/3691992.3692061"},{"key":"e_1_3_2_1_142_1","volume-title":"Llama: Open and Efficient Foundation Language Models. arXiv preprint arXiv:2302.13971","author":"Touvron Hugo","year":"2023","unstructured":"Hugo Touvron, Thibaut Lavril, Gautier Izacard, Xavier Martinet, Marie-Anne Lachaux, Timoth\u00e9e Lacroix, Baptiste Rozi\u00e8re, Naman Goyal, Eric Hambro, Faisal Azhar, et al., 2023a. Llama: Open and Efficient Foundation Language Models. arXiv preprint arXiv:2302.13971 (2023)."},{"key":"e_1_3_2_1_143_1","unstructured":"Hugo Touvron Louis Martin Kevin Stone Peter Albert Amjad Almahairi Yasmine Babaei Nikolay Bashlykov Soumya Batra Prajjwal Bhargava Shruti Bhosale et al. 2023b. Llama 2: Open Foundation and Fine-tuned Chat Models. arXiv preprint arXiv:2307.09288 (2023)."},{"key":"e_1_3_2_1_144_1","doi-asserted-by":"publisher","DOI":"10.1145\/3020078.3021744"},{"key":"e_1_3_2_1_145_1","volume-title":"Proceedings of the Advances in Neural Information Processing Systems 30 (NeurIPS).","author":"Vaswani Ashish","year":"2017","unstructured":"Ashish Vaswani, Noam Shazeer, Niki Parmar, Jakob Uszkoreit, Llion Jones, Aidan N Gomez, \u0141 ukasz Kaiser, and Illia Polosukhin. 2017. Attention is All You Need. In Proceedings of the Advances in Neural Information Processing Systems 30 (NeurIPS)."},{"key":"e_1_3_2_1_146_1","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA57654.2024.00033"},{"key":"e_1_3_2_1_147_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICPPW.2012.39"},{"key":"e_1_3_2_1_148_1","doi-asserted-by":"publisher","DOI":"10.1109\/ISSCC42615.2023.10067395"},{"key":"e_1_3_2_1_149_1","volume-title":"A Technical Overview of the Oracle Exadata Database Machine and Exadata Storage Server. Oracle White Paper","author":"Weiss Ronald","year":"2012","unstructured":"Ronald Weiss. 2012. A Technical Overview of the Oracle Exadata Database Machine and Exadata Storage Server. Oracle White Paper (2012)."},{"key":"e_1_3_2_1_150_1","doi-asserted-by":"publisher","DOI":"10.1145\/3543622.3573185"},{"key":"e_1_3_2_1_151_1","doi-asserted-by":"publisher","DOI":"10.14778\/3626292.3626303"},{"key":"e_1_3_2_1_152_1","doi-asserted-by":"publisher","DOI":"10.1145\/3490422.3502369"},{"key":"e_1_3_2_1_153_1","unstructured":"Xilinx. [n.d.]. AMD Kintex\u2122 UltraScale\u2122 FPGAs. https:\/\/www.amd.com\/en\/products\/adaptive-socs-and-fpgas\/fpga\/kintex-ultrascale-plus.html. Accessed: 2025-12-12."},{"key":"e_1_3_2_1_154_1","unstructured":"Xilinx. [n.d.]. Xilinx OpenCL Extension. https:\/\/xilinx.github.io\/XRT\/master\/html\/opencl_extension.html. Accessed: 2025-12-12."},{"key":"e_1_3_2_1_155_1","volume-title":"Proceedings of the 2022 International Symposium on High-Performance Computer Architecture (HPCA).","author":"Xiong W.","unstructured":"W. Xiong, L. Ke, D. Jankov, M. Kounavis, X. Wang, E. Northup, J. Yang, B. Acun, C. Wu, P. Peter Tang, G. Edward Suh, X. Zhang, and H. S. Lee. 2022. SecNDP: Secure Near-Data Processing with Untrusted Memory. In Proceedings of the 2022 International Symposium on High-Performance Computer Architecture (HPCA)."},{"key":"e_1_3_2_1_156_1","doi-asserted-by":"publisher","DOI":"10.1109\/SC41406.2024.00100"},{"key":"e_1_3_2_1_157_1","volume-title":"Orca: A Distributed Serving System for Transformer-Based Generative Models. In In Proceedings of the 16th USENIX Symposium on Operating Systems Design and Implementation (OSDI).","author":"Yu Gyeong-In","year":"2022","unstructured":"Gyeong-In Yu, Joo Seong Jeong, Geon-Woo Kim, Soojeong Kim, and Byung-Gon Chun. 2022. Orca: A Distributed Serving System for Transformer-Based Generative Models. In In Proceedings of the 16th USENIX Symposium on Operating Systems Design and Implementation (OSDI)."},{"key":"e_1_3_2_1_158_1","doi-asserted-by":"publisher","DOI":"10.1109\/MICRO61859.2024.00108"},{"key":"e_1_3_2_1_159_1","doi-asserted-by":"publisher","DOI":"10.1109\/MICRO61859.2024.00105"},{"key":"e_1_3_2_1_160_1","doi-asserted-by":"publisher","DOI":"10.1109\/PACT.2015.43"},{"key":"e_1_3_2_1_161_1","volume-title":"Proceedings of the 22nd USENIX Conference on File and Storage Technologies (FAST).","author":"Zhang Jian","year":"2024","unstructured":"Jian Zhang, Yujie Ren, Marie Nguyen, Changwoo Min, and Sudarsun Kannan. 2024. OmniCache: Collaborative Caching for Near-storage Accelerators. In Proceedings of the 22nd USENIX Conference on File and Storage Technologies (FAST)."},{"key":"e_1_3_2_1_162_1","volume-title":"Xi Victoria Lin, et al","author":"Zhang Susan","year":"2022","unstructured":"Susan Zhang, Stephen Roller, Naman Goyal, Mikel Artetxe, Moya Chen, Shuohui Chen, Christopher Dewan, Mona Diab, Xian Li, Xi Victoria Lin, et al., 2022. Opt: Open Pre-Trained Transformer Language Models. arXiv preprint arXiv:2205.01068 (2022)."},{"key":"e_1_3_2_1_163_1","volume-title":"Proceedings of the Advances in Neural Information Processing Systems 36 (NeurIPS).","author":"Zhang Zhenyu","year":"2023","unstructured":"Zhenyu Zhang, Ying Sheng, Tianyi Zhou, Tianlong Chen, Lianmin Zheng, Ruisi Cai, Zhao Song, Yuandong Tian, Christopher R\u00e9, Clark Barrett, Zhangyang ''Atlas'' Wang, and Beidi Chen. 2023. H2O: Heavy-Hitter Oracle for Efficient Generative Inference of Large Language Models. In Proceedings of the Advances in Neural Information Processing Systems 36 (NeurIPS)."},{"key":"e_1_3_2_1_164_1","volume-title":"HeteGen: Heterogeneous Parallel Inference for Large Language Models on Resource-Constrained Devices. arXiv preprint arXiv:2403.01164","author":"Zhao Xuanlei","year":"2024","unstructured":"Xuanlei Zhao, Bin Jia, Haotian Zhou, Ziming Liu, Shenggan Cheng, and Yang You. 2024. HeteGen: Heterogeneous Parallel Inference for Large Language Models on Resource-Constrained Devices. arXiv preprint arXiv:2403.01164 (2024)."},{"key":"e_1_3_2_1_165_1","volume-title":"Atom: Low-bit quantization for efficient and accurate llm serving. arXiv preprint arXiv:2310.19102","author":"Zhao Yilong","year":"2023","unstructured":"Yilong Zhao, Chien-Yu Lin, Kan Zhu, Zihao Ye, Lequn Chen, Size Zheng, Luis Ceze, Arvind Krishnamurthy, Tianqi Chen, and Baris Kasikci. 2023. Atom: Low-bit quantization for efficient and accurate llm serving. arXiv preprint arXiv:2310.19102 (2023)."},{"key":"e_1_3_2_1_166_1","volume-title":"Proceedings of the Advances in Neural Information Processing Systems 36 (NeurIPS).","author":"Zheng Lianmin","year":"2023","unstructured":"Lianmin Zheng, Wei-Lin Chiang, Ying Sheng, Siyuan Zhuang, Zhanghao Wu, Yonghao Zhuang, Zi Lin, Zhuohan Li, Dacheng Li, Eric Xing, Hao Zhang, Joseph E Gonzalez, and Ion Stoica. 2023. Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena. In Proceedings of the Advances in Neural Information Processing Systems 36 (NeurIPS)."},{"key":"e_1_3_2_1_167_1","volume-title":"DistServe: Disaggregating Prefill and Decoding for Goodput-optimized Large Language Model Serving. arXiv preprint arXiv:2401.09670","author":"Zhong Yinmin","year":"2024","unstructured":"Yinmin Zhong, Shengyu Liu, Junda Chen, Jianbo Hu, Yibo Zhu, Xuanzhe Liu, Xin Jin, and Hao Zhang. 2024. DistServe: Disaggregating Prefill and Decoding for Goodput-optimized Large Language Model Serving. arXiv preprint arXiv:2401.09670 (2024)."},{"key":"e_1_3_2_1_168_1","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA53966.2022.00082"}],"event":{"name":"ASPLOS '26: 31st ACM International Conference on Architectural Support for Programming Languages and Operating Systems","location":"Pittsburgh PA USA","sponsor":["SIGOPS ACM Special Interest Group on Operating Systems","SIGPLAN ACM Special Interest Group on Programming Languages","SIGARCH ACM Special Interest Group on Computer Architecture","SIGBED ACM Special Interest Group on Embedded Systems"]},"container-title":["Proceedings of the 31st ACM International Conference on Architectural Support for Programming Languages and Operating Systems, Volume 2"],"original-title":[],"deposited":{"date-parts":[[2026,3,15]],"date-time":"2026-03-15T13:59:07Z","timestamp":1773583147000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3779212.3790119"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,3,22]]},"references-count":168,"alternative-id":["10.1145\/3779212.3790119","10.1145\/3779212"],"URL":"https:\/\/doi.org\/10.1145\/3779212.3790119","relation":{},"subject":[],"published":{"date-parts":[[2026,3,22]]},"assertion":[{"value":"2026-03-22","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}