{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,11]],"date-time":"2026-07-11T05:12:24Z","timestamp":1783746744213,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":33,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,7,13]],"date-time":"2026-07-13T00:00:00Z","timestamp":1783900800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,7,13]]},"DOI":"10.1145\/3806645.3807596","type":"proceedings-article","created":{"date-parts":[[2026,7,11]],"date-time":"2026-07-11T04:21:11Z","timestamp":1783743671000},"page":"445-456","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Scaling Attention Beyond GPUs for LLM Inference"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0000-6550-7484","authenticated-orcid":false,"given":"Weishu","family":"Deng","sequence":"first","affiliation":[{"name":"The University of Texas at Arlington, Arlington, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-9087-9693","authenticated-orcid":false,"given":"Yujie","family":"Yang","sequence":"additional","affiliation":[{"name":"The University of Texas at Arlington, Arlington, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-8731-3478","authenticated-orcid":false,"given":"Peiran","family":"Du","sequence":"additional","affiliation":[{"name":"The University of Texas at Arlington, Arlington, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-7303-3595","authenticated-orcid":false,"given":"Lingfeng","family":"Xiang","sequence":"additional","affiliation":[{"name":"University of Texas at Arlington, Arlington, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-3653-7173","authenticated-orcid":false,"given":"Zhen","family":"Lin","sequence":"additional","affiliation":[{"name":"University of Texas at Arlington, Arlington, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4098-6260","authenticated-orcid":false,"given":"Chen","family":"Zhong","sequence":"additional","affiliation":[{"name":"University of texas at arlington, Arlington, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0688-6394","authenticated-orcid":false,"given":"Faraz","family":"Ahmed","sequence":"additional","affiliation":[{"name":"Hewlett Packard Labs, Milpitas, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3408-9050","authenticated-orcid":false,"given":"Lianjie","family":"Cao","sequence":"additional","affiliation":[{"name":"Hewlett Packard Enterprise, Milpitas, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4594-8164","authenticated-orcid":false,"given":"Puneet","family":"Sharma","sequence":"additional","affiliation":[{"name":"Hewlett Packard Labs, Milpitas, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1681-9008","authenticated-orcid":false,"given":"Song","family":"Jiang","sequence":"additional","affiliation":[{"name":"University of Texas at Arlington, Arlington, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-5062-4710","authenticated-orcid":false,"given":"Hui","family":"Lu","sequence":"additional","affiliation":[{"name":"The University of Texas at Arlington, Arlington, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2133-4363","authenticated-orcid":false,"given":"Jia","family":"Rao","sequence":"additional","affiliation":[{"name":"The University of Texas at Arlington, Arlington, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,7,13]]},"reference":[{"key":"e_1_3_3_2_2_2","unstructured":"Yushi Bai Xin Lv Jiajie Zhang Hongchang Lyu Jiankai Tang Zhidian Huang Zhengxiao Du Xiao Liu Aohan Zeng Lei Hou et\u00a0al. 2023. Longbench: A bilingual multitask benchmark for long context understanding. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2308.14508 (2023)."},{"key":"e_1_3_3_2_3_2","unstructured":"Iz Beltagy Matthew\u00a0E Peters and Arman Cohan. 2020. Longformer: The long-document transformer. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2004.05150 (2020)."},{"key":"e_1_3_3_2_4_2","doi-asserted-by":"crossref","unstructured":"Sid Black Stella Biderman Eric Hallahan Quentin Anthony Leo Gao Laurence Golding Horace He Connor Leahy Kyle McDonell Jason Phang et\u00a0al. 2022. Gpt-neox-20b: An open-source autoregressive language model. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2204.06745 (2022).","DOI":"10.18653\/v1\/2022.bigscience-1.9"},{"key":"e_1_3_3_2_5_2","unstructured":"Sumit\u00a0Kumar Dam Choong\u00a0Seon Hong Yu Qiao and Chaoning Zhang. 2024. A complete survey on llm-based ai chatbots. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2406.16937 (2024)."},{"key":"e_1_3_3_2_6_2","doi-asserted-by":"crossref","unstructured":"Tri Dao Dan Fu Stefano Ermon Atri Rudra and Christopher R\u00e9. 2022. Flashattention: Fast and memory-efficient exact attention with io-awareness. Advances in Neural Information Processing Systems 35 (2022) 16344\u201316359.","DOI":"10.52202\/068431-1189"},{"key":"e_1_3_3_2_7_2","doi-asserted-by":"crossref","unstructured":"Guhao Feng Bohang Zhang Yuntian Gu Haotian Ye Di He and Liwei Wang. 2024. Towards revealing the mystery behind chain of thought: a theoretical perspective. Advances in Neural Information Processing Systems 36 (2024).","DOI":"10.52202\/075280-3100"},{"key":"e_1_3_3_2_8_2","unstructured":"Nikita Kitaev \u0141ukasz Kaiser and Anselm Levskaya. 2020. Reformer: The efficient transformer. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2001.04451 (2020)."},{"key":"e_1_3_3_2_9_2","doi-asserted-by":"publisher","DOI":"10.1145\/3600006.3613165"},{"key":"e_1_3_3_2_10_2","first-page":"155","volume-title":"18th USENIX Symposium on Operating Systems Design and Implementation (OSDI 24)","author":"Lee Wonbeom","year":"2024","unstructured":"Wonbeom Lee, Jungi Lee, Junghwan Seo, and Jaewoong Sim. 2024. { InfiniGen} : Efficient generative inference of large language models with dynamic { KV} cache management. In 18th USENIX Symposium on Operating Systems Design and Implementation (OSDI 24). 155\u2013172."},{"key":"e_1_3_3_2_11_2","first-page":"19274","volume-title":"International Conference on Machine Learning","author":"Leviathan Yaniv","year":"2023","unstructured":"Yaniv Leviathan, Matan Kalman, and Yossi Matias. 2023. Fast inference from transformers via speculative decoding. In International Conference on Machine Learning. PMLR, 19274\u201319286."},{"key":"e_1_3_3_2_12_2","unstructured":"Chaofan Lin Jiaming Tang Shuo Yang Hanshuo Wang Tian Tang Boyu Tian Ion Stoica Song Han and Mingyu Gao. 2025. Twilight: Adaptive Attention Sparsity with Hierarchical Top-p Pruning. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2502.02770 (2025)."},{"key":"e_1_3_3_2_13_2","unstructured":"Enzhe Lu Zhejun Jiang Jingyuan Liu Yulun Du Tao Jiang Chao Hong Shaowei Liu Weiran He Enming Yuan Yuzhi Wang et\u00a0al. 2025. MoBA: Mixture of Block Attention for Long-Context LLMs. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2502.13189 (2025)."},{"key":"e_1_3_3_2_14_2","unstructured":"Xiurui Pan Endian Li Qiao Li Shengwen Liang Yizhou Shan Ke Zhou Yingwei Luo Xiaolin Wang and Jie Zhang. 2024. Instinfer: In-storage attention offloading for cost-effective long-context llm inference. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2409.04992 (2024)."},{"key":"e_1_3_3_2_15_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.emnlp-main.748"},{"key":"e_1_3_3_2_16_2","unstructured":"Baptiste Roziere Jonas Gehring Fabian Gloeckle Sten Sootla Itai Gat Xiaoqing\u00a0Ellen Tan Yossi Adi Jingyu Liu Romain Sauvestre Tal Remez et\u00a0al. 2023. Code llama: Open foundation models for code. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2308.12950 (2023)."},{"key":"e_1_3_3_2_17_2","first-page":"31094","volume-title":"International Conference on Machine Learning","author":"Sheng Ying","year":"2023","unstructured":"Ying Sheng, Lianmin Zheng, Binhang Yuan, Zhuohan Li, Max Ryabinin, Beidi Chen, Percy Liang, Christopher R\u00e9, Ion Stoica, and Ce Zhang. 2023. Flexgen: High-throughput generative inference of large language models with a single gpu. In International Conference on Machine Learning. PMLR, 31094\u201331116."},{"key":"e_1_3_3_2_18_2","unstructured":"Jiaming Tang Yilong Zhao Kan Zhu Guangxuan Xiao Baris Kasikci and Song Han. 2024. Quest: Query-Aware Sparsity for Efficient Long-Context LLM Inference. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2406.10774 (2024)."},{"key":"e_1_3_3_2_19_2","unstructured":"Hugo Touvron Thibaut Lavril Gautier Izacard Xavier Martinet Marie-Anne Lachaux Timoth\u00e9e Lacroix Baptiste Rozi\u00e8re Naman Goyal Eric Hambro Faisal Azhar et\u00a0al. 2023. Llama: Open and efficient foundation language models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2302.13971 (2023)."},{"key":"e_1_3_3_2_20_2","first-page":"1","volume-title":"Proceedings of the 63rd ACM\/IEEE Design Automation Conference (DAC \u201926)","author":"Wang Shihao","year":"2026","unstructured":"Shihao Wang, Xiangyu Zou, Wen Xia, and Hao Hu. 2026. An Ultra-fast and Lossless Floating-Point Compressor for LLM Inference Systems. In Proceedings of the 63rd ACM\/IEEE Design Automation Conference (DAC \u201926). ACM, 1\u20137."},{"key":"e_1_3_3_2_21_2","doi-asserted-by":"crossref","unstructured":"Jason Wei Xuezhi Wang Dale Schuurmans Maarten Bosma Fei Xia Ed Chi Quoc\u00a0V Le Denny Zhou et\u00a0al. 2022. Chain-of-thought prompting elicits reasoning in large language models. Advances in neural information processing systems 35 (2022) 24824\u201324837.","DOI":"10.52202\/068431-1800"},{"key":"e_1_3_3_2_22_2","unstructured":"Thomas Wolf Lysandre Debut Victor Sanh Julien Chaumond Clement Delangue Anthony Moi Pierric Cistac Tim Rault R\u00e9mi Louf Morgan Funtowicz et\u00a0al. 2019. Huggingface\u2019s transformers: State-of-the-art natural language processing. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1910.03771 (2019)."},{"key":"e_1_3_3_2_23_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICDE60146.2024.00378"},{"key":"e_1_3_3_2_24_2","unstructured":"Guangxuan Xiao Yuandong Tian Beidi Chen Song Han and Mike Lewis. 2023. Efficient streaming language models with attention sinks. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2309.17453 (2023)."},{"key":"e_1_3_3_2_25_2","unstructured":"Ruyi Xu Guangxuan Xiao Haofeng Huang Junxian Guo and Song Han. 2025. XAttention: Block Sparse Attention with Antidiagonal Scoring. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2503.16428 (2025)."},{"key":"e_1_3_3_2_26_2","doi-asserted-by":"crossref","unstructured":"Wenhua Ye Xu Zhou Joey Zhou Cen Chen and Kenli Li. 2023. Accelerating attention mechanism on fpgas based on efficient reconfigurable systolic array. ACM Transactions on Embedded Computing Systems 22 6 (2023) 1\u201322.","DOI":"10.1145\/3549937"},{"key":"e_1_3_3_2_27_2","unstructured":"Zihao Ye Lequn Chen Ruihang Lai Yilong Zhao Size Zheng Junru Shao Bohan Hou Hongyi Jin Yifei Zuo Liangsheng Yin et\u00a0al. 2024. Accelerating self-attentions for llm serving with flashinfer."},{"key":"e_1_3_3_2_28_2","unstructured":"Zihao Ye Ruihang Lai Bo-Ru Lu Chien-Yu Lin Size Zheng Lequn Chen Tianqi Chen and Luis Ceze. 2024. Cascade inference: Memory bandwidth efficient shared prefix batch decoding."},{"key":"e_1_3_3_2_29_2","unstructured":"Jingyang Yuan Huazuo Gao Damai Dai Junyu Luo Liang Zhao Zhengyan Zhang Zhenda Xie YX Wei Lean Wang Zhiping Xiao et\u00a0al. 2025. Native sparse attention: Hardware-aligned and natively trainable sparse attention. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2502.11089 (2025)."},{"key":"e_1_3_3_2_30_2","unstructured":"Manzil Zaheer Guru Guruganesh Kumar\u00a0Avinava Dubey Joshua Ainslie Chris Alberti Santiago Ontanon Philip Pham Anirudh Ravula Qifan Wang Li Yang et\u00a0al. 2020. Big bird: Transformers for longer sequences. Advances in neural information processing systems 33 (2020) 17283\u201317297."},{"key":"e_1_3_3_2_31_2","unstructured":"Susan Zhang Stephen Roller Naman Goyal Mikel Artetxe Moya Chen Shuohui Chen Christopher Dewan Mona Diab Xian Li Xi\u00a0Victoria Lin et\u00a0al. 2022. Opt: Open pre-trained transformer language models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2205.01068 (2022)."},{"key":"e_1_3_3_2_32_2","doi-asserted-by":"crossref","unstructured":"Zhenyu Zhang Ying Sheng Tianyi Zhou Tianlong Chen Lianmin Zheng Ruisi Cai Zhao Song Yuandong Tian Christopher R\u00e9 Clark Barrett et\u00a0al. 2024. H2o: Heavy-hitter oracle for efficient generative inference of large language models. Advances in Neural Information Processing Systems 36 (2024).","DOI":"10.52202\/075280-1506"},{"key":"e_1_3_3_2_33_2","unstructured":"Lianmin Zheng Liangsheng Yin Zhiqiang Xie Jeff Huang Chuyue Sun Cody_Hao Yu Shiyi Cao Christos Kozyrakis Ion Stoica Joseph\u00a0E Gonzalez et\u00a0al. 2023. Efficiently Programming Large Language Models using SGLang. (2023)."},{"key":"e_1_3_3_2_34_2","doi-asserted-by":"crossref","unstructured":"Shuzhang Zhong Yanfan Sun Ling Liang Runsheng Wang Ru Huang and Meng Li. 2025. HybriMoE: Hybrid CPU-GPU Scheduling and Cache Management for Efficient MoE Inference. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2504.05897 (2025).","DOI":"10.1109\/DAC63849.2025.11133274"}],"event":{"name":"HPDC '26: 35th International Symposium on High-Performance Parallel and Distributed Computing","location":"Cleveland USA","acronym":"HPDC '26","sponsor":["SIGHPC ACM Special Interest Group on High Performance Computing, Special Interest Group on High Performance Computing","SIGARCH ACM Special Interest Group on Computer Architecture"]},"container-title":["Proceedings of the 35th International Symposium on High-Performance Parallel and Distributed Computing"],"original-title":[],"deposited":{"date-parts":[[2026,7,11]],"date-time":"2026-07-11T04:24:00Z","timestamp":1783743840000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3806645.3807596"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,13]]},"references-count":33,"alternative-id":["10.1145\/3806645.3807596","10.1145\/3806645"],"URL":"https:\/\/doi.org\/10.1145\/3806645.3807596","relation":{},"subject":[],"published":{"date-parts":[[2026,7,13]]},"assertion":[{"value":"2026-07-13","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}