{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,22]],"date-time":"2026-06-22T11:08:36Z","timestamp":1782126516984,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":63,"publisher":"ACM","license":[{"start":{"date-parts":[[2025,5,14]],"date-time":"2025-05-14T00:00:00Z","timestamp":1747180800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0\/"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,5,14]]},"DOI":"10.1145\/3713082.3730390","type":"proceedings-article","created":{"date-parts":[[2025,6,6]],"date-time":"2025-06-06T09:53:51Z","timestamp":1749203631000},"page":"127-135","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":6,"title":["Good things come in small packages: Should we build AI clusters with Lite-GPUs?"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-8406-7768","authenticated-orcid":false,"given":"Burcu","family":"Canakci","sequence":"first","affiliation":[{"name":"Microsoft Research, United Kingdom"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4277-1802","authenticated-orcid":false,"given":"Junyi","family":"Liu","sequence":"additional","affiliation":[{"name":"Microsoft Research, United Kingdom"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1649-2612","authenticated-orcid":false,"given":"Xingbo","family":"Wu","sequence":"additional","affiliation":[{"name":"Microsoft Research, United Kingdom"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-5661-0007","authenticated-orcid":false,"given":"Nathana\u00ebl","family":"Cheriere","sequence":"additional","affiliation":[{"name":"Microsoft Research, United Kingdom"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1939-5690","authenticated-orcid":false,"given":"Paolo","family":"Costa","sequence":"additional","affiliation":[{"name":"Microsoft Research, United Kingdom"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-5596-8962","authenticated-orcid":false,"given":"Sergey","family":"Legtchenko","sequence":"additional","affiliation":[{"name":"Microsoft Research, United Kingdom"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-1194-2958","authenticated-orcid":false,"given":"Dushyanth","family":"Narayanan","sequence":"additional","affiliation":[{"name":"Microsoft Research, United Kingdom"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-5936-6895","authenticated-orcid":false,"given":"Antony","family":"Rowstron","sequence":"additional","affiliation":[{"name":"Microsoft Research, United Kingdom"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,6,6]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"[n.d.]. NVIDIA GB200 NVL72. https:\/\/www.nvidia.com\/en-us\/data-center\/gb200-nvl72\/. Accessed: 2025-01-15."},{"key":"e_1_3_2_1_2_1","volume-title":"Hot Chips 2024 Conference Proceedings. https:\/\/hc2024","year":"2020","unstructured":"2024. Hot Chips 2024 Conference Proceedings. https:\/\/hc2024.hotchips.org\/assets\/program\/conference\/day1\/HotChips%20-%202024-08-26.pdf Accessed: 2025-01-16."},{"key":"e_1_3_2_1_3_1","unstructured":"Marah Abdin Jyoti Aneja Hany Awadalla Ahmed Awadallah Ammar Ahmad Awan Nguyen Bach Amit Bahree Arash Bakhtiari Jianmin Bao Harkirat Behl Alon Benhaim Misha Bilenko Johan Bjorck S\u00e9bastien Bubeck Martin Cai Qin Cai Vishrav Chaudhary Dong Chen Dongdong Chen Weizhu Chen Yen-Chun Chen Yi-Ling Chen Hao Cheng Parul Chopra Xiyang Dai Matthew Dixon Ronen Eldan Victor Fragoso Jianfeng Gao Mei Gao Min Gao Amit Garg Allie Del Giorno Abhishek Goswami Suriya Gunasekar Emman Haider Junheng Hao Russell J. Hewett Wenxiang Hu Jamie Huynh Dan Iter Sam Ade Jacobs Mojan Javaheripi Xin Jin Nikos Karampatziakis Piero Kauffmann Mahoud Khademi Dongwoo Kim Young Jin Kim Lev Kurilenko James R. Lee Yin Tat Lee Yuanzhi Li Yunsheng Li Chen Liang Lars Liden Xihui Lin Zeqi Lin Ce Liu Liyuan Liu Mengchen Liu Weishung Liu Xiaodong Liu Chong Luo Piyush Madan Ali Mahmoudzadeh David Majercak Matt Mazzola Caio C\u00e9sar Teodoro Mendes Arindam Mitra Hardik Modi Anh Nguyen Brandon Norick Barun Patra Daniel Perez-Becker Thomas Portet Reid Pryzant Heyang Qin Marko Radmilac Liliang Ren Gustavo de Rosa Corby Rosset Sambudha Roy Olatunji Ruwase Olli Saarikivi Amin Saied Adil Salim Michael Santacroce Shital Shah Ning Shang Hiteshi Sharma Yelong Shen Swadheen Shukla Xia Song Masahiro Tanaka Andrea Tupini Praneetha Vaddamanu Chunyu Wang Guanhua Wang Lijuan Wang Shuohang Wang Xin Wang Yu Wang Rachel Ward Wen Wen Philipp Witte Haiping Wu Xiaoxia Wu Michael Wyatt Bin Xiao Can Xu Jiahang Xu Weijian Xu Jilong Xue Sonali Yadav Fan Yang Jianwei Yang Yifan Yang Ziyi Yang Donghan Yu Lu Yuan Chenruidong Zhang Cyril Zhang Jianwen Zhang Li Lyna Zhang Yi Zhang Yue Zhang Yunan Zhang and Xiren Zhou. 2024. Phi-3 Technical Report: A Highly Capable Language Model Locally on Your Phone. arXiv:2404.14219 [cs.CL] https:\/\/arxiv.org\/abs\/2404.14219"},{"key":"e_1_3_2_1_4_1","volume-title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills. arXiv:2308.16369 [cs.LG] https:\/\/arxiv.org\/abs\/2308.16369","author":"Agrawal Amey","year":"2023","unstructured":"Amey Agrawal, Ashish Panwar, Jayashree Mohan, Nipun Kwatra, Bhargav S. Gulavani, and Ramachandran Ramjee. 2023. SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills. arXiv:2308.16369 [cs.LG] https:\/\/arxiv.org\/abs\/2308.16369"},{"key":"e_1_3_2_1_5_1","volume-title":"Cheng Li, Du Li, Elton Zheng, Jeff Rasley, Shaden Smith, Olatunji Ruwase, and Yuxiong He.","author":"Aminabadi Reza Yazdani","year":"2022","unstructured":"Reza Yazdani Aminabadi, Samyam Rajbhandari, Minjia Zhang, Ammar Ahmad Awan, Cheng Li, Du Li, Elton Zheng, Jeff Rasley, Shaden Smith, Olatunji Ruwase, and Yuxiong He. 2022. DeepSpeed Inference: Enabling Efficient Inference of Transformer Models at Unprecedented Scale. arXiv:2207.00032 [cs.LG] https:\/\/arxiv.org\/abs\/2207.00032"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1145\/3387514.3406221"},{"key":"e_1_3_2_1_7_1","unstructured":"Tom B. Brown Benjamin Mann Nick Ryder Melanie Subbiah Jared Kaplan Prafulla Dhariwal Arvind Neelakantan Pranav Shyam Girish Sastry Amanda Askell Sandhini Agarwal Ariel Herbert-Voss Gretchen Krueger Tom Henighan Rewon Child Aditya Ramesh Daniel M. Ziegler Jeffrey Wu Clemens Winter Christopher Hesse Mark Chen Eric Sigler Mateusz Litwin Scott Gray Benjamin Chess Jack Clark Christopher Berner Sam McCandlish Alec Radford Ilya Sutskever and Dario Amodei. 2020. Language Models are Few-Shot Learners. arXiv:2005.14165 [cs.CL] https:\/\/arxiv.org\/abs\/2005.14165"},{"key":"e_1_3_2_1_8_1","unstructured":"Cerebras. [n.d.]. Wafer Scale Engine 3. https:\/\/www.cerebras.ai\/chip. [Accessed 17-04-2025]."},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1145\/3694715.3695970"},{"key":"e_1_3_2_1_10_1","unstructured":"Google Cloud. 2025. TPU v6e. https:\/\/cloud.google.com\/tpu\/docs\/v6e Accessed: 2025-01-15."},{"key":"e_1_3_2_1_11_1","unstructured":"NVIDIA Corporation. 2022. NVIDIA H100 Tensor Core GPU Architecture Overview. https:\/\/resources.nvidia.com\/en-us-tensor-core\/gtc22-whitepaper-hopper Accessed: 2025-01-10."},{"key":"e_1_3_2_1_12_1","unstructured":"NVIDIA Corporation. 2025. NVIDIA H100 NVL GPU. https:\/\/www.nvidia.com\/content\/dam\/en-zz\/Solutions\/Data-Center\/h100\/PB-11773-001_v01.pdf Accessed: 2025-01-15."},{"key":"e_1_3_2_1_13_1","first-page":"16344","article-title":"Flashattention: Fast and memory-efficient exact attention with io-awareness","volume":"35","author":"Dao Tri","year":"2022","unstructured":"Tri Dao, Dan Fu, Stefano Ermon, Atri Rudra, and Christopher R\u00e9. 2022. Flashattention: Fast and memory-efficient exact attention with io-awareness. Advances in Neural Information Processing Systems 35 (2022), 16344--16359.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_14_1","unstructured":"DeepSeek-AI Aixin Liu Bei Feng Bing Xue Bingxuan Wang Bochao Wu Chengda Lu Chenggang Zhao Chengqi Deng Chenyu Zhang Chong Ruan Damai Dai Daya Guo Dejian Yang Deli Chen Dongjie Ji Erhang Li Fangyun Lin Fucong Dai Fuli Luo Guangbo Hao Guanting Chen Guowei Li H. Zhang Han Bao Hanwei Xu Haocheng Wang Haowei Zhang Honghui Ding Huajian Xin Huazuo Gao Hui Li Hui Qu J. L. Cai Jian Liang Jianzhong Guo Jiaqi Ni Jiashi Li Jiawei Wang Jin Chen Jingchang Chen Jingyang Yuan Junjie Qiu Junlong Li Junxiao Song Kai Dong Kai Hu Kaige Gao Kang Guan Kexin Huang Kuai Yu Lean Wang Lecong Zhang Lei Xu Leyi Xia Liang Zhao Litong Wang Liyue Zhang Meng Li Miaojun Wang Mingchuan Zhang Minghua Zhang Minghui Tang Mingming Li Ning Tian Pan-pan Huang Peiyi Wang Peng Zhang Qiancheng Wang Qihao Zhu Qinyu Chen Qiushi Du R. J. Chen R. L. Jin Ruiqi Ge Ruisong Zhang Ruizhe Pan Runji Wang Runxin Xu Ruoyu Zhang Ruyi Chen S. S. Li Shanghao Lu Shangyan Zhou Shanhuang Chen Shaoqing Wu Shengfeng Ye Shengfeng Ye Shirong Ma Shiyu Wang Shuang Zhou Shuiping Yu Shunfeng Zhou Shuting Pan T. Wang Tao Yun Tian Pei Tianyu Sun W. L. Xiao Wangding Zeng Wanjia Zhao Wei An Wen Liu Wenfeng Liang Wenjun Gao Wenqin Yu Wentao Zhang X. Q. Li Xiangyue Jin Xianzu Wang Xiao Bi Xiaodong Liu Xiaohan Wang Xiaojin Shen Xiaokang Chen Xiaokang Zhang Xiaosha Chen Xiaotao Nie Xiaowen Sun Xiaoxiang Wang Xin Cheng Xin Liu Xin Xie Xingchao Liu Xingkai Yu Xinnan Song Xinxia Shan Xinyi Zhou Xinyu Yang Xinyuan Li Xuecheng Su Xuheng Lin Y. K. Li Y. Q. Wang Y. X. Wei Y. X. Zhu Yang Zhang Yanhong Xu Yanhong Xu Yanping Huang Yao Li Yao Zhao Yaofeng Sun Yaohui Li Yaohui Wang Yi Yu Yi Zheng Yichao Zhang Yifan Shi Yiliang Xiong Ying He Ying Tang Yishi Piao Yisong Wang Yixuan Tan Yiyang Ma Yiyuan Liu Yongqiang Guo Yu Wu Yuan Ou Yuchen Zhu Yuduan Wang Yue Gong Yuheng Zou Yujia He Yukun Zha Yunfan Xiong Yunxian Ma Yuting Yan Yuxiang Luo Yuxiang You Yuxuan Liu Yuyang Zhou Z. F. Wu Z. Z. Ren Zehui Ren Zhangli Sha Zhe Fu Zhean Xu Zhen Huang Zhen Zhang Zhenda Xie Zhengyan Zhang Zhewen Hao Zhibin Gou Zhicheng Ma Zhigang Yan Zhihong Shao Zhipeng Xu Zhiyu Wu Zhongyu Zhang Zhuoshu Li Zihui Gu Zijia Zhu Zijun Liu Zilin Li Ziwei Xie Ziyang Song Ziyi Gao and Zizheng Pan. 2025. DeepSeek-V3 Technical Report. arXiv:2412.19437 [cs.CL] https:\/\/arxiv.org\/abs\/2412.19437"},{"key":"e_1_3_2_1_15_1","unstructured":"Rob Van der Wijngaart and Fred Oh. 2022. Boosting Application Performance with GPU Memory Prefetching. https:\/\/developer.nvidia.com\/blog\/boosting-application-performance-with-gpu-memory-prefetching\/ Accessed: 2025-01-15."},{"key":"e_1_3_2_1_16_1","unstructured":"Hugging Face. [n. d.]. Llama 3.1 - 405B 70B & 8B with multilinguality and long context. https:\/\/huggingface.co\/blog\/llama31. [Accessed 15-04-2025]."},{"key":"e_1_3_2_1_17_1","unstructured":"Yichao Fu Siqi Zhu Runlong Su Aurick Qiao Ion Stoica and Hao Zhang. 2024. Efficient LLM Scheduling by Learning to Rank. arXiv:2408.15792 [cs.LG] https:\/\/arxiv.org\/abs\/2408.15792"},{"key":"e_1_3_2_1_18_1","volume-title":"IFTLE 607: Why Nvidia's Blackwell is Having Issues with TSMC CoWoS-L Technology. 3DInCites","author":"Garrou Phil","year":"2024","unstructured":"Phil Garrou. 2024. IFTLE 607: Why Nvidia's Blackwell is Having Issues with TSMC CoWoS-L Technology. 3DInCites (2024). https:\/\/www.3dincites.com\/2024\/10\/iftle-607-why-nvidias-blackwell-is-having-issues-with-tsmc-cowos-l-technology\/ Accessed: 2025-01-14."},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1109\/JSSC.1972.1052898"},{"key":"e_1_3_2_1_20_1","unstructured":"Horace He. 2024. Strangely Matrix Multiplications on GPUs Run Faster When Given \"Predictable\" Data! https:\/\/www.thonking.ai\/p\/strangely-matrix-multiplications Accessed: 2025-01-15."},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1115\/IPACK2022-97416"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1145\/3696348.3696872"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1109\/MCAS.2024.3349669"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1145\/3579371.3589350"},{"key":"e_1_3_2_1_25_1","volume-title":"Joseph E. Gonzalez, Hao Zhang, and Ion Stoica.","author":"Kwon Woosuk","year":"2023","unstructured":"Woosuk Kwon, Zhuohan Li, Siyuan Zhuang, Ying Sheng, Lianmin Zheng, Cody Hao Yu, Joseph E. Gonzalez, Hao Zhang, and Ion Stoica. 2023. Efficient Memory Management for Large Language Model Serving with PagedAttention. arXiv:2309.06180 [cs.LG] https:\/\/arxiv.org\/abs\/2309.06180"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1145\/3579370.3594777"},{"key":"e_1_3_2_1_27_1","unstructured":"Hao Liu Matei Zaharia and Pieter Abbeel. 2023. Ring Attention with Blockwise Transformers for Near-Infinite Context. arXiv:2310.01889 [cs.CL] https:\/\/arxiv.org\/abs\/2310.01889"},{"key":"e_1_3_2_1_28_1","volume-title":"How We'll Reach a 1 Trillion Transistor GPU","author":"Liu Mark","year":"2024","unstructured":"Mark Liu and H.-S. Philip Wong. 2024. How We'll Reach a 1 Trillion Transistor GPU. IEEE Spectrum (March 2024). https:\/\/spectrum.ieee.org\/trillion-transistor-gpu Accessed: 2025-01-15."},{"key":"e_1_3_2_1_29_1","volume-title":"Automation & Test in Europe Conference & Exhibition (DATE). IEEE, 142--145","author":"Loh Gabriel H","year":"2021","unstructured":"Gabriel H Loh, Samuel Naffziger, and Kevin Lepak. 2021. Understanding chiplets today to anticipate future integration opportunities and limits. In 2021 Design, Automation & Test in Europe Conference & Exhibition (DATE). IEEE, 142--145."},{"key":"e_1_3_2_1_30_1","volume-title":"Memory Disaggregation: Advances and Open Challenges. arXiv:2305.03943 [cs.DC] https:\/\/arxiv.org\/abs\/2305.03943","author":"Maruf Hasan Al","year":"2023","unstructured":"Hasan Al Maruf and Mosharaf Chowdhury. 2023. Memory Disaggregation: Advances and Open Challenges. arXiv:2305.03943 [cs.DC] https:\/\/arxiv.org\/abs\/2305.03943"},{"key":"e_1_3_2_1_31_1","unstructured":"Meta. [n. d.]. Introducing Llama 3.1: Our most capable models to date. https:\/\/ai.meta.com\/blog\/meta-llama-3-1\/. [Accessed 15-04-2025]."},{"key":"e_1_3_2_1_32_1","unstructured":"Meta. 2025. Meta Llama on Hugging Face. https:\/\/huggingface.co\/meta-llama. Accessed: 2025-01-13."},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1145\/3620665.3640411"},{"key":"e_1_3_2_1_34_1","unstructured":"Shervin Minaee Tomas Mikolov Narjes Nikzad Meysam Chenaghlu Richard Socher Xavier Amatriain and Jianfeng Gao. 2024. Large Language Models: A Survey. arXiv:2402.06196 [cs.CL] https:\/\/arxiv.org\/abs\/2402.06196"},{"key":"e_1_3_2_1_35_1","volume-title":"Co-packaged datacenter optics: Opportunities and challenges. IET optoelectronics 15, 2","author":"Minkenberg Cyriel","year":"2021","unstructured":"Cyriel Minkenberg, Rajagopal Krishnaswamy, Aaron Zilkie, and David Nelson. 2021. Co-packaged datacenter optics: Opportunities and challenges. IET optoelectronics 15, 2 (2021), 77--91."},{"key":"e_1_3_2_1_36_1","volume-title":"d.]. Die Yield Calculator","author":"Elite Moore","unstructured":"Moore Elite. [n. d.]. Die Yield Calculator. http:\/\/cloud.mooreelite.com\/tools\/die-yield-calculator\/index.html. Accessed: 2025-01-15."},{"key":"e_1_3_2_1_37_1","volume-title":"Nvidia Blackwell GPUs Allegedly Delayed Due to Design Flaws. Tom's Hardware","author":"Morales Jowi","year":"2024","unstructured":"Jowi Morales. 2024. Nvidia Blackwell GPUs Allegedly Delayed Due to Design Flaws. Tom's Hardware (2024). https:\/\/www.tomshardware.com\/pc-components\/gpus\/nvidia-blackwell-gpus-allegedly-delayed-due-to-design-flaws Accessed: 2025-01-10."},{"key":"e_1_3_2_1_38_1","unstructured":"NVIDIA. [n.d.]. NVIDIA Announces Spectrum-X Photonics Co-Packaged Optics Networking Switches to Scale AI Factories to Millions of GPUs. https:\/\/nvidianews.nvidia.com\/news\/nvidia-spectrum-x-co-packaged-optics-networking-switches-ai-factories. [Accessed 15-04-2025]."},{"key":"e_1_3_2_1_39_1","unstructured":"NVIDIA. 2025. NVIDIA Project DIGITS. https:\/\/www.nvidia.com\/en-us\/project-digits\/. Accessed: 2025-01-14."},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA59077.2024.00019"},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1145\/3617232.3624853"},{"key":"e_1_3_2_1_42_1","volume-title":"2024 USENIX Annual Technical Conference (USENIX ATC 24)","author":"Qiu Haoran","year":"2024","unstructured":"Haoran Qiu, Weichao Mao, Archit Patke, Shengkun Cui, Saurabh Jha, Chen Wang, Hubertus Franke, Zbigniew Kalbarczyk, Tamer Ba\u015far, and Ravishankar K Iyer. 2024. Power-aware Deep Learning Model Serving with {&mu;-Serve}. In 2024 USENIX Annual Technical Conference (USENIX ATC 24). 75--93."},{"key":"e_1_3_2_1_43_1","volume-title":"Flashattention-3: Fast and accurate attention with asynchrony and low-precision. arXiv preprint arXiv:2407.08608","author":"Shah Jay","year":"2024","unstructured":"Jay Shah, Ganesh Bikshandi, Ying Zhang, Vijay Thakkar, Pradeep Ramani, and Tri Dao. 2024. Flashattention-3: Fast and accurate attention with asynchrony and low-precision. arXiv preprint arXiv:2407.08608 (2024)."},{"key":"e_1_3_2_1_44_1","unstructured":"Mohammad Shoeybi Mostofa Patwary Raul Puri Patrick LeGresley Jared Casper and Bryan Catanzaro. 2020. Megatron-LM: Training Multi-Billion Parameter Language Models Using Model Parallelism. arXiv:1909.08053 [cs.CL] https:\/\/arxiv.org\/abs\/1909.08053"},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1145\/3694715.3695964"},{"key":"e_1_3_2_1_46_1","volume-title":"Towards Greener LLMs: Bringing Energy-Efficiency to the Forefront of LLM Inference. arXiv preprint arXiv:2403.20306","author":"Stojkovic Jovan","year":"2024","unstructured":"Jovan Stojkovic, Esha Choukse, Chaojie Zhang, Inigo Goiri, and Josep Torrellas. 2024. Towards Greener LLMs: Bringing Energy-Efficiency to the Forefront of LLM Inference. arXiv preprint arXiv:2403.20306 (2024)."},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA59077.2024.00040"},{"key":"e_1_3_2_1_48_1","volume-title":"Fault-tolerant Generative LLM Serving. In Proceedings of the 41st International Conference on Machine Learning (Proceedings of Machine Learning Research","volume":"46771","author":"Strati Foteini","year":"2024","unstructured":"Foteini Strati, Sara Mcallister, Amar Phanishayee, Jakub Tarnawski, and Ana Klimovic. 2024. D\u00e9j\u00e0Vu: KV-cache Streaming for Fast, Fault-tolerant Generative LLM Serving. In Proceedings of the 41st International Conference on Machine Learning (Proceedings of Machine Learning Research, Vol. 235), Ruslan Salakhutdinov, Zico Kolter, Katherine Heller, Adrian Weller, Nuria Oliver, Jonathan Scarlett, and Felix Berkenkamp (Eds.). PMLR, 46745--46771. https:\/\/proceedings.mlr.press\/v235\/strati24a.html"},{"key":"e_1_3_2_1_49_1","volume-title":"Llumnix: Dynamic Scheduling for Large Language Model Serving. arXiv preprint arXiv:2406.03243","author":"Sun Biao","year":"2024","unstructured":"Biao Sun, Ziming Huang, Hanyu Zhao, Wencong Xiao, Xinyi Zhang, Yong Li, and Wei Lin. 2024. Llumnix: Dynamic Scheduling for Large Language Model Serving. arXiv preprint arXiv:2406.03243 (2024)."},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"publisher","DOI":"10.1007\/s12200-022-00055-y"},{"key":"e_1_3_2_1_51_1","unstructured":"Tianqi Tang and Yuan Xie. 2022. Cost-Aware Exploration for Chiplet-Based Architecture with Advanced Packaging Technologies. arXiv:2206.07308 [cs.AR] https:\/\/arxiv.org\/abs\/2206.07308"},{"key":"e_1_3_2_1_52_1","volume-title":"Nvidia faces order delays as Blackwell chips overheat. digwatch","author":"Team DW","year":"2025","unstructured":"DW Team. 2025. Nvidia faces order delays as Blackwell chips overheat. digwatch (2025). https:\/\/dig.watch\/updates\/nvidia-faces-order-delays-as-blackwell-chips-overheat Accessed: 2025-01-14."},{"key":"e_1_3_2_1_53_1","doi-asserted-by":"publisher","DOI":"10.1109\/66.536118"},{"key":"e_1_3_2_1_54_1","unstructured":"The Mac Observer. 2025. What is Apple Neural Engine? https:\/\/www.macobserver.com\/tips\/what-is-apple-neural-engine\/ Accessed: 2025-01-14."},{"key":"e_1_3_2_1_55_1","volume-title":"NVIDIA Blackwell Platform: Advancing Generative AI and Accelerated Computing. In 2024 IEEE Hot Chips 36 Symposium (HCS). IEEE Computer Society, 1--33","author":"Tirumala Ajay","year":"2024","unstructured":"Ajay Tirumala and Raymond Wong. 2024. NVIDIA Blackwell Platform: Advancing Generative AI and Accelerated Computing. In 2024 IEEE Hot Chips 36 Symposium (HCS). IEEE Computer Society, 1--33."},{"key":"e_1_3_2_1_56_1","unstructured":"Ashish Vaswani Noam Shazeer Niki Parmar Jakob Uszkoreit Llion Jones Aidan N. Gomez Lukasz Kaiser and Illia Polosukhin. 2023. Attention Is All You Need. arXiv:1706.03762 [cs.CL] https:\/\/arxiv.org\/abs\/1706.03762"},{"key":"e_1_3_2_1_57_1","doi-asserted-by":"publisher","DOI":"10.1145\/1498765.1498785"},{"key":"e_1_3_2_1_58_1","volume-title":"Empowering 1000 tokens\/second on-device llm prefilling with mllm-npu. arXiv preprint arXiv:2407.05858","author":"Xu Daliang","year":"2024","unstructured":"Daliang Xu, Hao Zhang, Liming Yang, Ruiqi Liu, Gang Huang, Mengwei Xu, and Xuanzhe Liu. 2024. Empowering 1000 tokens\/second on-device llm prefilling with mllm-npu. arXiv preprint arXiv:2407.05858 (2024)."},{"key":"e_1_3_2_1_59_1","unstructured":"Daliang Xu Hao Zhang Liming Yang Ruiqi Liu Gang Huang Mengwei Xu and Xuanzhe Liu. 2024. Fast On-device LLM Inference with NPUs. arXiv:2407.05858 [cs.AI] https:\/\/arxiv.org\/abs\/2407.05858"},{"key":"e_1_3_2_1_60_1","unstructured":"Jiajun Xu Zhiyuan Li Wei Chen Qun Wang Xin Gao Qi Cai and Ziyuan Ling. 2024. On-Device Language Models: A Comprehensive Review. arXiv:2409.00088 [cs.CL] https:\/\/arxiv.org\/abs\/2409.00088"},{"key":"e_1_3_2_1_61_1","doi-asserted-by":"publisher","DOI":"10.1145\/3442442.3452055"},{"key":"e_1_3_2_1_62_1","doi-asserted-by":"publisher","DOI":"10.1109\/MNET.005.2100604"},{"key":"e_1_3_2_1_63_1","unstructured":"Yinmin Zhong Shengyu Liu Junda Chen Jianbo Hu Yibo Zhu Xuanzhe Liu Xin Jin and Hao Zhang. 2024. DistServe: Disaggregating Prefill and Decoding for Goodput-optimized Large Language Model Serving. arXiv:2401.09670 [cs.DC] https:\/\/arxiv.org\/abs\/2401.09670"}],"event":{"name":"HOTOS '25: Workshop on Hot Topics in Operating Systems","location":"Banff AB Canada","acronym":"HOTOS '25","sponsor":["SIGOPS ACM Special Interest Group on Operating Systems"]},"container-title":["Proceedings of the Workshop on Hot Topics in Operating Systems"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3713082.3730390","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3713082.3730390","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,29]],"date-time":"2025-08-29T16:47:45Z","timestamp":1756486065000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3713082.3730390"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,5,14]]},"references-count":63,"alternative-id":["10.1145\/3713082.3730390","10.1145\/3713082"],"URL":"https:\/\/doi.org\/10.1145\/3713082.3730390","relation":{},"subject":[],"published":{"date-parts":[[2025,5,14]]},"assertion":[{"value":"2025-06-06","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}