{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T20:28:39Z","timestamp":1782937719405,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":34,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,9,16]],"date-time":"2024-09-16T00:00:00Z","timestamp":1726444800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,9,16]]},"DOI":"10.1145\/3688351.3689164","type":"proceedings-article","created":{"date-parts":[[2024,9,16]],"date-time":"2024-09-16T06:19:50Z","timestamp":1726467590000},"page":"91-103","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":19,"title":["TwinPilots: A New Computing Paradigm for GPU-CPU Parallel LLM Inference"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0005-7670-5247","authenticated-orcid":false,"given":"Chengye","family":"Yu","sequence":"first","affiliation":[{"name":"Chinese University of Hong Kong, Hong Kong, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7030-1990","authenticated-orcid":false,"given":"Tianyu","family":"Wang","sequence":"additional","affiliation":[{"name":"Shenzhen University, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2173-2847","authenticated-orcid":false,"given":"Zili","family":"Shao","sequence":"additional","affiliation":[{"name":"Chinese University of Hong Kong, Hong Kong, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2914-8749","authenticated-orcid":false,"given":"Linjie","family":"Zhu","sequence":"additional","affiliation":[{"name":"Sangfor Technologies Inc, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-7695-5834","authenticated-orcid":false,"given":"Xu","family":"Zhou","sequence":"additional","affiliation":[{"name":"Sangfor Technologies Inc, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1681-9008","authenticated-orcid":false,"given":"Song","family":"Jiang","sequence":"additional","affiliation":[{"name":"University of Texas at Arlington, Arlington, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2024,9,16]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"AMD. 2024. AMD EPYC 7002 Series Processors. https:\/\/www.amd.com\/en\/products\/processors\/server\/epyc\/7002-series.html"},{"key":"e_1_3_2_1_2_1","unstructured":"Rob Armstrong. 2022. CUDA Toolkit 12.0 Released for General Availability. https:\/\/developer.nvidia.com\/blog\/cuda-toolkit-12-0-released-for-general-availability\/"},{"key":"e_1_3_2_1_3_1","volume-title":"PipeSwitch: Fast Pipelined Context Switching for Deep Learning Applications. In 14th USENIX Symposium on Operating Systems Design and Implementation (OSDI 20)","author":"Bai Zhihao","year":"2020","unstructured":"Zhihao Bai, Zhen Zhang, Yibo Zhu, and Xin Jin. 2020. PipeSwitch: Fast Pipelined Context Switching for Deep Learning Applications. In 14th USENIX Symposium on Operating Systems Design and Implementation (OSDI 20). USENIX Association, 499--514. https:\/\/www.usenix.org\/conference\/osdi20\/presentation\/bai"},{"key":"e_1_3_2_1_4_1","unstructured":"BentoML. 2024. OpenLLM. https:\/\/github.com\/bentoml\/OpenLLM"},{"key":"e_1_3_2_1_5_1","first-page":"240","article-title":"2023. Palm: Scaling language modeling with pathways","volume":"24","author":"Chowdhery Aakanksha","year":"2023","unstructured":"Aakanksha Chowdhery, Sharan Narang, Jacob Devlin, Maarten Bosma, Gaurav Mishra, Adam Roberts, Paul Barham, Hyung Won Chung, Charles Sutton, Sebastian Gehrmann, et al. 2023. Palm: Scaling language modeling with pathways. Journal of Machine Learning Research 24, 240 (2023), 1--113.","journal-title":"Journal of Machine Learning Research"},{"key":"e_1_3_2_1_6_1","unstructured":"Machine Learning Compilation. 2024. MLC LLM. https:\/\/llm.mlc.ai\/"},{"key":"e_1_3_2_1_7_1","unstructured":"Georgi Gerganov. 2024. llama.cpp. https:\/\/github.com\/ggerganov\/llama.cpp"},{"key":"e_1_3_2_1_8_1","unstructured":"HuggingFace. 2024. Large Language Model Text Generation Inference. https:\/\/huggingface.co\/docs\/text-generation-inference\/index"},{"key":"e_1_3_2_1_9_1","unstructured":"Intel. 2022. Intel\u00ae Performance Counter Monitor - A Better Way to Measure CPU Utilization. https:\/\/www.intel.com\/content\/www\/us\/en\/developer\/articles\/tool\/performance-counter-monitor.html"},{"key":"e_1_3_2_1_10_1","unstructured":"Intel. 2024. Intel\u00ae Advanced Vector Extensions. https:\/\/software.intel.com\/en-us\/isa-extensions\/intel-avx"},{"key":"e_1_3_2_1_11_1","unstructured":"Intel. 2024. Intel\u00ae oneAPI Math Kernel Library (oneMKL). https:\/\/www.intel.com\/content\/www\/us\/en\/developer\/tools\/oneapi\/onemkl.html"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1145\/3552326.3567508"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.ngt-1.25"},{"key":"e_1_3_2_1_14_1","volume-title":"Joseph E. Gonzalez, Hao Zhang, and Ion Stoica.","author":"Kwon Woosuk","year":"2023","unstructured":"Woosuk Kwon, Zhuohan Li, Siyuan Zhuang, Ying Sheng, Lianmin Zheng, Cody Hao Yu, Joseph E. Gonzalez, Hao Zhang, and Ion Stoica. 2023. Efficient Memory Management for Large Language Model Serving with PagedAttention. arXiv:cs.LG\/2309.06180"},{"key":"e_1_3_2_1_15_1","volume-title":"Approximation algorithms for scheduling unrelated parallel machines. Mathematical programming 46","author":"Lenstra Jan Karel","year":"1990","unstructured":"Jan Karel Lenstra, David B Shmoys, and \u00c9va Tardos. 1990. Approximation algorithms for scheduling unrelated parallel machines. Mathematical programming 46 (1990), 259--271."},{"key":"e_1_3_2_1_16_1","unstructured":"Microsoft. 2024. DeepSpeed Model Implementations for Inference (MII). https:\/\/www.microsoft.com\/en-us\/research\/project\/deepspeed\/deepspeed-mii\/"},{"key":"e_1_3_2_1_17_1","unstructured":"NVIDIA. 2024. NVIDIA A10 GPU Accelerator. https:\/\/www.nvidia. cn\/content\/dam\/en-zz\/Solutions\/Data-Center\/a10\/pdf\/A10-ProductBrief.pdf"},{"key":"e_1_3_2_1_18_1","unstructured":"NVIDIA. 2024. NVIDIA A100 Tensor Core GPU. https:\/\/www.nvidia.cn\/data-center\/a100\/"},{"key":"e_1_3_2_1_19_1","unstructured":"OpenMP. 2024. OpenMP. https:\/\/www.openmp.org\/"},{"key":"e_1_3_2_1_20_1","unstructured":"PyTorch. 2024. PyTorch Docs. https:\/\/pytorch.org\/"},{"key":"e_1_3_2_1_21_1","unstructured":"Alec Radford Jeffrey Wu Rewon Child David Luan Dario Amodei Ilya Sutskever et al. 2019. Language models are unsupervised multitask learners. OpenAI blog 1 8 (2019) 9."},{"key":"e_1_3_2_1_22_1","unstructured":"Bibhudatta Sahoo. 2013. Dynamic load balancing strategies in heterogeneous distributed system. Ph.D. Dissertation."},{"key":"e_1_3_2_1_23_1","volume-title":"Ray Serve: Scalable and Programmable Serving. https:\/\/docs.ray.io\/en\/latest\/serve\/index.html","author":"Serve Ray","year":"2024","unstructured":"Ray Serve. 2024. Ray Serve: Scalable and Programmable Serving. https:\/\/docs.ray.io\/en\/latest\/serve\/index.html"},{"key":"e_1_3_2_1_24_1","volume-title":"Proceedings of the 40th International Conference on Machine Learning","author":"Sheng Ying","year":"2023","unstructured":"Ying Sheng, Lianmin Zheng, Binhang Yuan, Zhuohan Li, Max Ryabinin, Beidi Chen, Percy Liang, Christopher R\u00e9, Ion Stoica, and Ce Zhang. 2023. FlexGen: High-Throughput Generative Inference of Large Language Models with a Single GPU. In Proceedings of the 40th International Conference on Machine Learning (Honolulu, Hawaii, USA) (ICML'23). JMLR.org, Article 1288, 23 pages."},{"key":"e_1_3_2_1_25_1","volume-title":"Megatron-LM: Training Multi-Billion Parameter Language Models Using Model Parallelism. CoRR abs\/1909.08053","author":"Shoeybi Mohammad","year":"2019","unstructured":"Mohammad Shoeybi, Mostofa Patwary, Raul Puri, Patrick LeGresley, Jared Casper, and Bryan Catanzaro. 2019. Megatron-LM: Training Multi-Billion Parameter Language Models Using Model Parallelism. CoRR abs\/1909.08053 (2019). arXiv:1909.08053 http:\/\/arxiv.org\/abs\/1909.08053"},{"key":"e_1_3_2_1_26_1","volume-title":"Jamie Hall, Noam Shazeer, Apoorv Kulshreshtha, Heng-Tze Cheng, Alicia Jin, Taylor Bos, Leslie Baker, Yu Du, et al.","author":"Thoppilan Romal","year":"2022","unstructured":"Romal Thoppilan, Daniel De Freitas, Jamie Hall, Noam Shazeer, Apoorv Kulshreshtha, Heng-Tze Cheng, Alicia Jin, Taylor Bos, Leslie Baker, Yu Du, et al. 2022. Lamda: Language models for dialog applications. arXiv preprint arXiv:2201.08239 (2022)."},{"key":"e_1_3_2_1_27_1","volume-title":"Llama: Open and efficient foundation language models. arXiv preprint arXiv:2302.13971","author":"Touvron Hugo","year":"2023","unstructured":"Hugo Touvron, Thibaut Lavril, Gautier Izacard, Xavier Martinet, Marie-Anne Lachaux, Timoth\u00e9e Lacroix, Baptiste Rozi\u00e8re, Naman Goyal, Eric Hambro, Faisal Azhar, et al. 2023. Llama: Open and efficient foundation language models. arXiv preprint arXiv:2302.13971 (2023)."},{"key":"e_1_3_2_1_28_1","volume-title":"Stone et al","author":"Touvron Hugo","year":"2023","unstructured":"Hugo Touvron, Louis Martin, and Kevin R. Stone et al. 2023. Llama 2: Open Foundation and Fine-Tuned Chat Models. ArXiv abs\/2307.09288 (2023). https:\/\/api.semanticscholar.org\/CorpusID:259950998"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.5555\/3295222.3295349"},{"key":"e_1_3_2_1_30_1","volume-title":"Tree of thoughts: Deliberate problem solving with large language models. Advances in Neural Information Processing Systems 36","author":"Yao Shunyu","year":"2024","unstructured":"Shunyu Yao, Dian Yu, Jeffrey Zhao, Izhak Shafran, Tom Griffiths, Yuan Cao, and Karthik Narasimhan. 2024. Tree of thoughts: Deliberate problem solving with large language models. Advances in Neural Information Processing Systems 36 (2024)."},{"key":"e_1_3_2_1_31_1","volume-title":"Prompting large language model for machine translation: A case study. arXiv preprint arXiv:2301.07069","author":"Zhang Biao","year":"2023","unstructured":"Biao Zhang, Barry Haddow, and Alexandra Birch. 2023. Prompting large language model for machine translation: A case study. arXiv preprint arXiv:2301.07069 (2023)."},{"key":"e_1_3_2_1_32_1","volume-title":"Xi Victoria Lin, et al","author":"Zhang Susan","year":"2022","unstructured":"Susan Zhang, Stephen Roller, Naman Goyal, Mikel Artetxe, Moya Chen, Shuohui Chen, Christopher Dewan, Mona Diab, Xian Li, Xi Victoria Lin, et al. 2022. Opt: Open pre-trained transformer language models. arXiv preprint arXiv:2205.01068 (2022)."},{"key":"e_1_3_2_1_33_1","volume-title":"Benchmarking large language models for news summarization. arXiv preprint arXiv:2301.13848","author":"Zhang Tianyi","year":"2023","unstructured":"Tianyi Zhang, Faisal Ladhak, Esin Durmus, Percy Liang, Kathleen McKeown, and Tatsunori B Hashimoto. 2023. Benchmarking large language models for news summarization. arXiv preprint arXiv:2301.13848 (2023)."},{"key":"e_1_3_2_1_34_1","volume-title":"Rethinking positional encoding. arXiv preprint arXiv:2107.02561","author":"Zheng Jianqiao","year":"2021","unstructured":"Jianqiao Zheng, Sameera Ramasinghe, and Simon Lucey. 2021. Rethinking positional encoding. arXiv preprint arXiv:2107.02561 (2021)."}],"event":{"name":"SYSTOR '24: The 17th ACM International Systems and Storage Conference","location":"Virtual Israel","acronym":"SYSTOR '24","sponsor":["SIGOPS ACM Special Interest Group on Operating Systems","Technion Israel Institute of Technology"]},"container-title":["Proceedings of the 17th ACM International Systems and Storage Conference on ZZZ"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3688351.3689164","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3688351.3689164","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,26]],"date-time":"2025-08-26T19:52:41Z","timestamp":1756237961000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3688351.3689164"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,9,16]]},"references-count":34,"alternative-id":["10.1145\/3688351.3689164","10.1145\/3688351"],"URL":"https:\/\/doi.org\/10.1145\/3688351.3689164","relation":{},"subject":[],"published":{"date-parts":[[2024,9,16]]},"assertion":[{"value":"2024-09-16","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}