{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,7]],"date-time":"2026-07-07T15:38:30Z","timestamp":1783438710893,"version":"3.54.6"},"publisher-location":"New York, NY, USA","reference-count":28,"publisher":"ACM","funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62302175"],"award-info":[{"award-number":["62302175"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,12]]},"DOI":"10.1145\/3725783.3764409","type":"proceedings-article","created":{"date-parts":[[2025,10,9]],"date-time":"2025-10-09T17:50:12Z","timestamp":1760032212000},"page":"54-60","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":1,"title":["HyperGen: Optimizing Generative Inference with Long Prompts for Resource-Constrained Systems"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0007-9966-3398","authenticated-orcid":false,"given":"Lingwen","family":"Gong","sequence":"first","affiliation":[{"name":"Huazhong University of Science and Technology, Wuhan, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-9358-2376","authenticated-orcid":false,"given":"Kaixin","family":"Liu","sequence":"additional","affiliation":[{"name":"Huazhong University of Science and Technology, Wuhan, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3582-0792","authenticated-orcid":false,"given":"Xiaolu","family":"Li","sequence":"additional","affiliation":[{"name":"Huazhong University of Science and Technology, Wuhan, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5311-5782","authenticated-orcid":false,"given":"Shujie","family":"Han","sequence":"additional","affiliation":[{"name":"Northwestern Polytechnical University, Xi'an, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4501-4364","authenticated-orcid":false,"given":"Patrick P. C.","family":"Lee","sequence":"additional","affiliation":[{"name":"The Chinese University of Hong Kong, Hong Kong, Hong Kong"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1265-7141","authenticated-orcid":false,"given":"Yuchong","family":"Hu","sequence":"additional","affiliation":[{"name":"Huazhong University of Science and Technology, Wuhan, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4674-6006","authenticated-orcid":false,"given":"Dan","family":"Feng","sequence":"additional","affiliation":[{"name":"Huazhong University of Science and Technology, Wuhan, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,10,11]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Jared Kaplan, Harri Edwards, Yuri Burda, Nicholas Joseph, Greg Brockman, et al. Evaluating large language models trained on code. arXiv preprint arXiv:2107.03374","author":"Chen Mark","year":"2021","unstructured":"Mark Chen, Jerry Tworek, Heewoo Jun, Qiming Yuan, Henrique Ponde De Oliveira Pinto, Jared Kaplan, Harri Edwards, Yuri Burda, Nicholas Joseph, Greg Brockman, et al. Evaluating large language models trained on code. arXiv preprint arXiv:2107.03374, 2021."},{"key":"e_1_3_2_1_2_1","first-page":"113155","volume-title":"Advances in Neural Information Processing Systems","author":"Chen Renze","unstructured":"Renze Chen, Zhuofeng Wang, Beiquan Cao, Tong Wu, Size Zheng, Xiuhong Li, Xuechao Wei, Shengen Yan, Meng Li, and Yun Liang. Ark-Vale: Efficient Generative LLM Inference with Recallable Key-Value Eviction. In A. Globerson, L. Mackey, D. Belgrave, A. Fan, U. Paquet, J. Tomczak, and C. Zhang, editors, Advances in Neural Information Processing Systems, volume 37, pages 113134\u2013113155. Curran Associates, Inc., 2024."},{"key":"e_1_3_2_1_3_1","first-page":"201","volume-title":"Gang Chen. IMPRESS: An Importance-Informed Multi-Tier Prefix KV Storage System for Large Language Model Inference. In 23rd USENIX Conference on File and Storage Technologies (FAST 25)","author":"Chen Weijian","year":"2025","unstructured":"Weijian Chen, Shuibing He, Haoyang Qu, Ruidong Zhang, Siling Yang, Ping Chen, Yi Zheng, Baoxing Huai, and Gang Chen. IMPRESS: An Importance-Informed Multi-Tier Prefix KV Storage System for Large Language Model Inference. In 23rd USENIX Conference on File and Storage Technologies (FAST 25), pages 187\u2013201, 2025."},{"key":"e_1_3_2_1_4_1","volume-title":"RetroInfer: A Vector-Storage Approach for Scalable Long-Context LLM Inference. arXiv preprint arXiv:2505.02922","author":"Chen Yaoqi","year":"2025","unstructured":"Yaoqi Chen, Jinkai Zhang, Baotong Lu, Qianxi Zhang, Chengruidong Zhang, Jingjia Luo, Di Liu, Huiqiang Jiang, Qi Chen, Jing Liu, et al. RetroInfer: A Vector-Storage Approach for Scalable Long-Context LLM Inference. arXiv preprint arXiv:2505.02922, 2025."},{"key":"e_1_3_2_1_5_1","volume-title":"NetGPT: A native-AI network architecture beyond provisioning personalized generative services. arXiv preprint arXiv:2307.06148","author":"Chen Yuxuan","year":"2023","unstructured":"Yuxuan Chen, Rongpeng Li, Zhifeng Zhao, Chenghui Peng, Jianjun Wu, Ekram Hossain, and Honggang Zhang. NetGPT: A native-AI network architecture beyond provisioning personalized generative services. arXiv preprint arXiv:2307.06148, 2023."},{"key":"e_1_3_2_1_6_1","volume-title":"Beidi Chen. Magic PIG: LSH Sampling for Efficient LLM Generation. In The Thirteenth International Conference on Learning Representations","author":"Chen Zhuoming","year":"2025","unstructured":"Zhuoming Chen, Ranajoy Sadhukhan, Zihao Ye, Yang Zhou, Jianyu Zhang, Niklas Nolte, Yuandong Tian, Matthijs Douze, Leon Bottou, Zhihao Jia, and Beidi Chen. Magic PIG: LSH Sampling for Efficient LLM Generation. In The Thirteenth International Conference on Learning Representations, 2025."},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.patrec.2018.02.009"},{"key":"e_1_3_2_1_8_1","volume-title":"Workshop on Efficient Systems for Foundation Models II @ ICML2024","author":"Fu Qichen","year":"2024","unstructured":"Qichen Fu, Minsik Cho, Thomas Merth, Sachin Mehta, Mohammad Rastegari, and Mahyar Najibi. LazyLLM: Dynamic token pruning for efficient long context LLM inference. In Workshop on Efficient Systems for Foundation Models II @ ICML2024, 2024."},{"key":"e_1_3_2_1_9_1","volume-title":"The Thirteenth International Conference on Learning Representations","author":"Hao Jitai","year":"2025","unstructured":"Jitai Hao, Yuke Zhu, Tian Wang, Jun Yu, Xin Xin, Bo Zheng, Zhaochun Ren, and Sheng Guo. OmniKV: Dynamic context selection for efficient long-context LLMs. In The Thirteenth International Conference on Learning Representations, 2025."},{"key":"e_1_3_2_1_10_1","volume-title":"Springer International Publishing","author":"Imambi Sagar","year":"2021","unstructured":"Sagar Imambi, Kolla Bhanu Prakash, and G. R. Kanagachidambaresan. PyTorch, pages 87\u2013104. Springer International Publishing, Cham, 2021."},{"key":"e_1_3_2_1_11_1","first-page":"172","volume-title":"Jaewoong Sim. Infini-Gen: Efficient Generative Inference of Large Language Models with Dynamic KV Cache Management. In 18th USENIX Symposium on Operating Systems Design and Implementation (OSDI 24)","author":"Lee Wonbeom","year":"2024","unstructured":"Wonbeom Lee, Jungi Lee, Junghwan Seo, and Jaewoong Sim. Infini-Gen: Efficient Generative Inference of Large Language Models with Dynamic KV Cache Management. In 18th USENIX Symposium on Operating Systems Design and Implementation (OSDI 24), pages 155\u2013172, Santa Clara, CA, July 2024. USENIX Association."},{"key":"e_1_3_2_1_12_1","volume-title":"ICML 2025 Workshop on Long-Context Foundation Models","author":"Luo Cheng","year":"2025","unstructured":"Cheng Luo, Zefan Cai, Hanshi Sun, Jinqi Xiao, Bo Yuan, Wen Xiao, Junjie Hu, Jiawei Zhao, Beidi Chen, and Anima Anandkumar. Head-Infer: Memory-Efficient LLM Inference by Head-wise Offloading. In ICML 2025 Workshop on Long-Context Foundation Models, 2025."},{"key":"e_1_3_2_1_13_1","volume-title":"https:\/\/www.llama.com\/","year":"2025","unstructured":"Meta. Llama. https:\/\/www.llama.com\/, 2025."},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.ijmedinf.2024.105474"},{"key":"e_1_3_2_1_15_1","volume-title":"The Eleventh International Conference on Learning Representations","author":"Nijkamp Erik","year":"2023","unstructured":"Erik Nijkamp, Bo Pang, Hiroaki Hayashi, Lifu Tu, Huan Wang, Yingbo Zhou, Silvio Savarese, and Caiming Xiong. CodeGen: An open large language model for code with multi-turn program synthesis. In The Eleventh International Conference on Learning Representations, 2023."},{"key":"e_1_3_2_1_16_1","volume-title":"OpenAI's gigantic GPT-3 hints at the limits of language models for AI. =https:\/\/www.zdnet.com\/article\/openais-gigantic-gpt-3-hints-at-the-limits-of-language-models-for-ai\/","author":"Ray Tiernan","year":"2020","unstructured":"Tiernan Ray. OpenAI's gigantic GPT-3 hints at the limits of language models for AI. =https:\/\/www.zdnet.com\/article\/openais-gigantic-gpt-3-hints-at-the-limits-of-language-models-for-ai\/, 2020. Accessed: 2025-05-23."},{"key":"e_1_3_2_1_17_1","volume-title":"Yossi Adi, Jingyu Liu, Romain Sauvestre, Tal Remez, et al. Code Llama: Open foundation models for code. arXiv preprint arXiv:2308.12950","author":"Roziere Baptiste","year":"2023","unstructured":"Baptiste Roziere, Jonas Gehring, Fabian Gloeckle, Sten Sootla, Itai Gat, Xiaoqing Ellen Tan, Yossi Adi, Jingyu Liu, Romain Sauvestre, Tal Remez, et al. Code Llama: Open foundation models for code. arXiv preprint arXiv:2308.12950, 2023."},{"key":"e_1_3_2_1_18_1","series-title":"Proceedings of Machine Learning Research","first-page":"31116","volume-title":"Proceedings of the 40th International Conference on Machine Learning","author":"Sheng Ying","unstructured":"Ying Sheng, Lianmin Zheng, Binhang Yuan, Zhuohan Li, Max Ryabinin, Beidi Chen, Percy Liang, Christopher Re, Ion Stoica, and Ce Zhang. FlexGen: High-Throughput generative inference of large language models with a single GPU. In Andreas Krause, Emma Brunskill, Kyunghyun Cho, Barbara Engelhardt, Sivan Sabato, and Jonathan Scarlett, editors, Proceedings of the 40th International Conference on Machine Learning, volume 202 of Proceedings of Machine Learning Research, pages 31094\u201331116. PMLR, 23\u201329 Jul 2023."},{"key":"e_1_3_2_1_19_1","first-page":"606","volume-title":"Proceedings of the ACM SIGOPS 30th Symposium on Operating Systems Principles, SOSP '24","author":"Song Yixin","year":"2024","unstructured":"Yixin Song, Zeyu Mi, Haotong Xie, and Haibo Chen. PowerInfer: Fast large language model serving with a consumer-grade gpu. In Proceedings of the ACM SIGOPS 30th Symposium on Operating Systems Principles, SOSP '24, pages 590\u2013606, New York, NY, USA, 2024. Association for Computing Machinery."},{"key":"e_1_3_2_1_20_1","volume-title":"Advances in Neural Information Processing Systems","author":"Vaswani Ashish","year":"2017","unstructured":"Ashish Vaswani, Noam Shazeer, Niki Parmar, Jakob Uszkoreit, Llion Jones, Aidan N Gomez, \u0141 ukasz Kaiser, and Illia Polosukhin. Attention is all you need. In I. Guyon, U. Von Luxburg, S. Bengio, H. Wallach, R. Fergus, S. Vishwanathan, and R. Garnett, editors, Advances in Neural Information Processing Systems, volume 30. Curran Associates, Inc., 2017."},{"key":"e_1_3_2_1_21_1","volume-title":"Privatelora for efficient privacy preserving LLM. arXiv preprint arXiv:2311.14030","author":"Wang Yiming","year":"2023","unstructured":"Yiming Wang, Yu Lin, Xiaodong Zeng, and Guannan Zhang. Privatelora for efficient privacy preserving LLM. arXiv preprint arXiv:2311.14030, 2023."},{"key":"e_1_3_2_1_22_1","volume-title":"Offsite-tuning: Transfer learning without full model. arXiv preprint arXiv:2302.04870","author":"Xiao Guangxuan","year":"2023","unstructured":"Guangxuan Xiao, Ji Lin, and Song Han. Offsite-tuning: Transfer learning without full model. arXiv preprint arXiv:2302.04870, 2023."},{"key":"e_1_3_2_1_23_1","volume-title":"Qwen3 technical report. arXiv preprint arXiv:2505.09388","author":"Yang An","year":"2025","unstructured":"An Yang, Anfeng Li, Baosong Yang, Beichen Zhang, Binyuan Hui, Bo Zheng, Bowen Yu, Chang Gao, Chengen Huang, Chenxu Lv, et al. Qwen3 technical report. arXiv preprint arXiv:2505.09388, 2025."},{"key":"e_1_3_2_1_24_1","first-page":"103","volume-title":"Proceedings of the 17th ACM International Systems and Storage Conference, SYSTOR '24","author":"Yu Chengye","year":"2024","unstructured":"Chengye Yu, Tianyu Wang, Zili Shao, Linjie Zhu, Xu Zhou, and Song Jiang. TwinPilots: A new computing paradigm for GPU-CPU parallel LLM inference. In Proceedings of the 17th ACM International Systems and Storage Conference, SYSTOR '24, pages 91\u2013103, New York, NY, USA, 2024. Association for Computing Machinery."},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"crossref","first-page":"100577","DOI":"10.1016\/j.smhl.2025.100577","article-title":"Dynamic fog computing for enhanced LLM execution in medical applications","volume":"36","author":"Zagar Philipp","year":"2025","unstructured":"Philipp Zagar, Vishnu Ravi, Lauren Aalami, Stephan Krusche, Oliver Aalami, and Paul Schmiedmayer. Dynamic fog computing for enhanced LLM execution in medical applications. Smart Health, 36:100577, 2025.","journal-title":"Smart Health"},{"key":"e_1_3_2_1_26_1","first-page":"34710","volume-title":"A. Oh","author":"Zhang Zhenyu","unstructured":"Zhenyu Zhang, Ying Sheng, Tianyi Zhou, Tianlong Chen, Lianmin Zheng, Ruisi Cai, Zhao Song, Yuandong Tian, Christopher R\u00e9, Clark Barrett, Zhangyang \"Atlas\" Wang, and Beidi Chen. H2o: Heavy-Hitter oracle for efficient generative inference of large language models. In A. Oh, T. Naumann, A. Globerson, K. Saenko, M. Hardt, and S. Levine, editors, Advances in Neural Information Processing Systems, volume 36, pages 34661\u201334710. Curran Associates, Inc., 2023."},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1109\/JPROC.2019.2918951"},{"key":"e_1_3_2_1_28_1","volume-title":"SpecOffload: Unlocking Latent GPU Capacity for LLM Inference on Resource-Constrained Devices. arXiv preprint arXiv:2505.10259","author":"Zhuge Xiangwen","year":"2025","unstructured":"Xiangwen Zhuge, Xu Shen, Zeyu Wang, Fan Dang, Xuan Ding, Danyang Li, Yahui Han, Tianxiang Hao, and Zheng Yang. SpecOffload: Unlocking Latent GPU Capacity for LLM Inference on Resource-Constrained Devices. arXiv preprint arXiv:2505.10259, 2025."}],"event":{"name":"APSys '25: 16th ACM SIGOPS Asia-Pacific Workshop on Systems","location":"Lotte Hotel World, Emerald Hall Seoul Republic of Korea","acronym":"APSys '25","sponsor":["SIGOPS ACM Special Interest Group on Operating Systems"]},"container-title":["Proceedings of the 16th ACM SIGOPS Asia-Pacific Workshop on Systems"],"original-title":[],"deposited":{"date-parts":[[2025,10,9]],"date-time":"2025-10-09T17:50:37Z","timestamp":1760032237000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3725783.3764409"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,11]]},"references-count":28,"alternative-id":["10.1145\/3725783.3764409","10.1145\/3725783"],"URL":"https:\/\/doi.org\/10.1145\/3725783.3764409","relation":{},"subject":[],"published":{"date-parts":[[2025,10,11]]},"assertion":[{"value":"2025-10-11","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}