{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,23]],"date-time":"2026-07-23T23:02:18Z","timestamp":1784847738189,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":45,"publisher":"ACM","license":[{"start":{"date-parts":[[2025,10,17]],"date-time":"2025-10-17T00:00:00Z","timestamp":1760659200000},"content-version":"vor","delay-in-days":0,"URL":"http:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"DOI":"10.13039\/100000001","name":"National Science Foundation","doi-asserted-by":"publisher","award":["CCF-2530337, CCF-2312741"],"award-info":[{"award-number":["CCF-2530337, CCF-2312741"]}],"id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100000028","name":"Semiconductor Research Corporation","doi-asserted-by":"publisher","id":[{"id":"10.13039\/100000028","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,18]]},"DOI":"10.1145\/3725843.3756062","type":"proceedings-article","created":{"date-parts":[[2025,10,17]],"date-time":"2025-10-17T17:19:56Z","timestamp":1760721596000},"page":"34-48","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":3,"title":["LongSight: Compute-Enabled Memory to Accelerate Large-Context LLMs via Sparse Attention"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0000-5862-6565","authenticated-orcid":false,"given":"Derrick","family":"Quinn","sequence":"first","affiliation":[{"name":"Cornell University, Ithaca, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-0460-8230","authenticated-orcid":false,"given":"E. Ezgi","family":"Y\u00fccel","sequence":"additional","affiliation":[{"name":"Cornell University, Ithaca, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1744-1393","authenticated-orcid":false,"given":"Jinkwon","family":"Kim","sequence":"additional","affiliation":[{"name":"Cornell University, Ithaca, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5451-5681","authenticated-orcid":false,"given":"Jos\u00e9 F.","family":"Mart\u00ednez","sequence":"additional","affiliation":[{"name":"Cornell University, Ithaca, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4622-2181","authenticated-orcid":false,"given":"Mohammad","family":"Alian","sequence":"additional","affiliation":[{"name":"Cornell University, Ithaca, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,10,17]]},"reference":[{"key":"e_1_3_3_2_2_2","doi-asserted-by":"crossref","unstructured":"Joshua Ainslie James Lee-Thorp Michiel de Jong Yury Zemlyanskiy Federico Lebr\u00f3n and Sumit Sanghai. 2023. GQA: Training Generalized Multi-Query Transformer Models from Multi-Head Checkpoints. arxiv:https:\/\/arXiv.org\/abs\/2305.13245\u00a0[cs.CL] https:\/\/arxiv.org\/abs\/2305.13245","DOI":"10.18653\/v1\/2023.emnlp-main.298"},{"key":"e_1_3_3_2_3_2","unstructured":"Iz Beltagy Matthew\u00a0E. Peters and Arman Cohan. 2020. Longformer: The Long-Document Transformer. arxiv:https:\/\/arXiv.org\/abs\/2004.05150\u00a0[cs.CL] https:\/\/arxiv.org\/abs\/2004.05150"},{"key":"e_1_3_3_2_4_2","unstructured":"NVIDIA\u00a0Developer Blog. 2025. Demystifying AI Inference Deployments for Trillion\u2011Parameter Large Language Models. Online; accessed 2025-06-21. https:\/\/developer.nvidia.com\/blog\/demystifying-ai-inference-deployments-for-trillion-parameter-large-language-models\/"},{"key":"e_1_3_3_2_5_2","unstructured":"Tom\u00a0B. Brown Benjamin Mann Nick Ryder Melanie Subbiah Jared Kaplan Prafulla Dhariwal Arvind Neelakantan Pranav Shyam Girish Sastry Amanda Askell Sandhini Agarwal Ariel Herbert-Voss Gretchen Krueger Tom Henighan Rewon Child Aditya Ramesh Daniel\u00a0M. Ziegler Jeffrey Wu Clemens Winter Christopher Hesse Mark Chen Eric Sigler Mateusz Litwin Scott Gray Benjamin Chess Jack Clark Christopher Berner Sam McCandlish Alec Radford Ilya Sutskever and Dario Amodei. 2020. Language Models are Few-Shot Learners. arxiv:https:\/\/arXiv.org\/abs\/2005.14165\u00a0[cs.CL] https:\/\/arxiv.org\/abs\/2005.14165"},{"key":"e_1_3_3_2_6_2","doi-asserted-by":"publisher","unstructured":"William\u00a0J. Dally Yatish Turakhia and Song Han. 2020. Domain-specific hardware accelerators. Commun. ACM 63 7 (2020) 48\u201357. 10.1145\/3361682","DOI":"10.1145\/3361682"},{"key":"e_1_3_3_2_7_2","volume-title":"Proceedings of the 31st Hot Chips Symposium (HC31)","author":"Devaux Fabrice","year":"2019","unstructured":"Fabrice Devaux. 2019. UPMEM Processing in Memory: DRAM is Becoming a True Processing Unit. In Proceedings of the 31st Hot Chips Symposium (HC31). Stanford, CA, USA. https:\/\/old.hotchips.org\/hc31\/HC31_1.4_UPMEM.FabriceDevaux.v2_1.pdf Accessed: November 23, 2024."},{"key":"e_1_3_3_2_8_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2011.5995432"},{"key":"e_1_3_3_2_9_2","unstructured":"Yufeng Gu Alireza Khadem Sumanth Umesh Ning Liang Xavier Servot Onur Mutlu Ravi Iyer and Reetuparna Das. 2025. PIM Is All You Need: A CXL-Enabled GPU-Free System for Large Language Model Inference. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2502.07578 (2025)."},{"key":"e_1_3_3_2_10_2","doi-asserted-by":"publisher","DOI":"10.1145\/3620666.3651380"},{"key":"e_1_3_3_2_11_2","unstructured":"Hewlett Packard Enterprise. 2025. HPE Superdome Flex 280 Interactive Demo. https:\/\/apps.kaonadn.net\/5185710160084992\/product.html#1\/199;C187. Accessed: 2025-04-09."},{"key":"e_1_3_3_2_12_2","unstructured":"Hewlett Packard Enterprise. 2025. HPE Superdome: Mission-Critical Servers. https:\/\/www.hpe.com\/us\/en\/servers\/superdome.html. Accessed: 2025-04-09."},{"key":"e_1_3_3_2_13_2","unstructured":"Coleman Hooper Sehoon Kim Hiva Mohammadzadeh Monishwaran Maheswaran June Paik Michael\u00a0W. Mahoney Kurt Keutzer and Amir Gholami. 2024. Squeezed Attention: Accelerating Long Context Length LLM Inference. arxiv:https:\/\/arXiv.org\/abs\/2411.09688\u00a0[cs.CL] https:\/\/arxiv.org\/abs\/2411.09688"},{"key":"e_1_3_3_2_14_2","doi-asserted-by":"publisher","unstructured":"Xinting Huang and Nora Hollenstein. 2023. Long-Range Language Modeling with Selective Cache. Findings of the Association for Computational Linguistics: EMNLP 2023 (December 2023) 4838\u20134858. 10.18653\/v1\/2023.findings-emnlp.321","DOI":"10.18653\/v1\/2023.findings-emnlp.321"},{"key":"e_1_3_3_2_15_2","doi-asserted-by":"publisher","unstructured":"Yoongu Kim Weikun Yang and Onur Mutlu. 2016. Ramulator: A Fast and Extensible DRAM Simulator. IEEE Computer Architecture Letters 15 1 (2016) 45\u201349. 10.1109\/LCA.2015.2414456","DOI":"10.1109\/LCA.2015.2414456"},{"key":"e_1_3_3_2_16_2","unstructured":"Nikita Kitaev \u0141ukasz Kaiser and Anselm Levskaya. 2020. Reformer: The Efficient Transformer. arxiv:https:\/\/arXiv.org\/abs\/2001.04451\u00a0[cs.LG] https:\/\/arxiv.org\/abs\/2001.04451"},{"key":"e_1_3_3_2_17_2","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA52012.2021.00013"},{"key":"e_1_3_3_2_18_2","unstructured":"Patrick Lewis Ethan Perez Aleksandra Piktus Fabio Petroni Vladimir Karpukhin Naman Goyal Heinrich K\u00fcttler Mike Lewis Wen tau Yih Tim Rockt\u00e4schel Sebastian Riedel and Douwe Kiela. 2021. Retrieval-Augmented Generation for Knowledge-Intensive NLP Tasks. arxiv:https:\/\/arXiv.org\/abs\/2005.11401\u00a0[cs.CL] https:\/\/arxiv.org\/abs\/2005.11401"},{"key":"e_1_3_3_2_19_2","doi-asserted-by":"publisher","DOI":"10.1145\/3575693.3578835"},{"key":"e_1_3_3_2_20_2","doi-asserted-by":"publisher","unstructured":"Shang Li Zhiyuan Yang Dhiraj Reddy Ankur Srivastava and Bruce Jacob. 2020. DRAMsim3: A Cycle-Accurate Thermal-Capable DRAM Simulator. IEEE Computer Architecture Letters 19 2 (2020) 106\u2013109. 10.1109\/LCA.2020.2973991","DOI":"10.1109\/LCA.2020.2973991"},{"key":"e_1_3_3_2_21_2","unstructured":"Yujun Lin Haotian Tang Shang Yang Zhekai Zhang Guangxuan Xiao Chuang Gan and Song Han. 2024. QServe: W4A8KV4 Quantization and System Co-design for Efficient LLM Serving. arxiv:https:\/\/arXiv.org\/abs\/2405.04532\u00a0[cs.CL] https:\/\/arxiv.org\/abs\/2405.04532"},{"key":"e_1_3_3_2_22_2","doi-asserted-by":"publisher","DOI":"10.1145\/3582016.3582063"},{"key":"e_1_3_3_2_23_2","unstructured":"Mistral AI team. 2023. Announcing Mistral 7B. https:\/\/mistral.ai\/news\/announcing-mistral-7b. Online; accessed 2025-06-19."},{"key":"e_1_3_3_2_24_2","volume-title":"Big Iron Will Always Drive Big Spending","author":"Morgan Timothy\u00a0Prickett","year":"2021","unstructured":"Timothy\u00a0Prickett Morgan. 2021. Big Iron Will Always Drive Big Spending. https:\/\/www.nextplatform.com\/2021\/09\/21\/big-iron-will-always-drive-big-spending\/ Accessed: 2025-04-09."},{"key":"e_1_3_3_2_25_2","unstructured":"CJ Newburn. 2024. GPUs as Data Access Engines. Conference presentation. https:\/\/files.futurememorystorage.com\/proceedings\/2024\/20240808_NETC-301-1_Newburn.pdf Presentation at the Flash Memory Summit (FMS)."},{"key":"e_1_3_3_2_26_2","unstructured":"NVIDIA. 2025. NIM LLMs Benchmarking: Performance. Online; accessed 2025-10-06. Benchmark latency and throughput numbers for Llama models via NVIDIA NIM."},{"key":"e_1_3_3_2_27_2","unstructured":"Maxwell Nye Anders\u00a0Johan Andreassen Guy Gur-Ari Henryk Michalewski Jacob Austin David Bieber David Dohan Aitor Lewkowycz Maarten Bosma David Luan Charles Sutton and Augustus Odena. 2021. Show Your Work: Scratchpads for Intermediate Computation with Language Models. arxiv:https:\/\/arXiv.org\/abs\/2112.00114\u00a0[cs.LG] https:\/\/arxiv.org\/abs\/2112.00114"},{"key":"e_1_3_3_2_28_2","doi-asserted-by":"publisher","unstructured":"OpenAI. 2025. gpt\u2011oss\u2011120b & gpt\u2011oss\u201120b Model Card. Online; accessed 2025\u201108\u201128. arXiv arXiv:2508.10925 (2025). 10.48550\/arXiv.2508.10925","DOI":"10.48550\/arXiv.2508.10925"},{"key":"e_1_3_3_2_29_2","unstructured":"OpenAI. 2025. Introducing Deep Research. https:\/\/openai.com\/index\/introducing-deep-research\/. Online; accessed 2025-04-11."},{"key":"e_1_3_3_2_30_2","doi-asserted-by":"publisher","DOI":"10.1145\/3620665.3640422"},{"key":"e_1_3_3_2_31_2","doi-asserted-by":"publisher","unstructured":"Jeongmin\u00a0Brian Park Vikram\u00a0Sharma Mailthody Zaid Qureshi and Wen-mei Hwu. 2024. Accelerating Sampling and Aggregation Operations in GNN Frameworks with GPU Initiated Direct Storage Accesses. Proc. VLDB Endow. 17 6 (Feb. 2024) 1227\u20131240. 10.14778\/3648160.3648166","DOI":"10.14778\/3648160.3648166"},{"key":"e_1_3_3_2_32_2","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA57654.2024.00078"},{"key":"e_1_3_3_2_33_2","doi-asserted-by":"publisher","unstructured":"Yeonhong Park Jake Hyun SangLyul Cho Bonggeun Sim and Jae\u00a0W. Lee. 2024. Any-Precision LLM: Low-Cost Deployment of Multiple Different-Sized LLMs. arXiv (2024). 10.48550\/arXiv.2402.10517 arxiv:https:\/\/arXiv.org\/abs\/2402.10517\u00a0[cs.LG]","DOI":"10.48550\/arXiv.2402.10517"},{"key":"e_1_3_3_2_34_2","unstructured":"Project Gutenberg. 1971. Project Gutenberg. https:\/\/www.gutenberg.org"},{"key":"e_1_3_3_2_35_2","doi-asserted-by":"publisher","DOI":"10.1145\/3695053.3731079"},{"key":"e_1_3_3_2_36_2","doi-asserted-by":"publisher","DOI":"10.1145\/3575693.3575748"},{"key":"e_1_3_3_2_37_2","doi-asserted-by":"publisher","unstructured":"Aaron Stillmaker and Bevan Baas. 2017. Scaling equations for the accurate prediction of CMOS device performance from 180nm to 7nm. Integration 58 (2017) 74\u201381. 10.1016\/j.vlsi.2017.02.002","DOI":"10.1016\/j.vlsi.2017.02.002"},{"key":"e_1_3_3_2_38_2","doi-asserted-by":"publisher","unstructured":"Gemma Team and Google DeepMind. 2025. Gemma 3 Technical Report. arXiv (12 March 2025). 10.48550\/arXiv.2503.19786 arxiv:https:\/\/arXiv.org\/abs\/2503.19786","DOI":"10.48550\/arXiv.2503.19786"},{"key":"e_1_3_3_2_39_2","unstructured":"Llama team. 2024. The Llama 3 Herd of Models. https:\/\/ai.meta.com\/research\/publications\/the-llama-3-herd-of-models\/"},{"key":"e_1_3_3_2_40_2","volume-title":"Advances in Neural Information Processing Systems","author":"Vaswani Ashish","year":"2017","unstructured":"Ashish Vaswani, Noam Shazeer, Niki Parmar, Jakob Uszkoreit, Llion Jones, Aidan\u00a0N Gomez, \u0141\u00a0ukasz Kaiser, and Illia Polosukhin. 2017. Attention is All you Need. In Advances in Neural Information Processing Systems , I.\u00a0Guyon, U.\u00a0Von Luxburg, S.\u00a0Bengio, H.\u00a0Wallach, R.\u00a0Fergus, S.\u00a0Vishwanathan, and R.\u00a0Garnett (Eds.), Vol.\u00a030. Curran Associates, Inc.https:\/\/proceedings.neurips.cc\/paper_files\/paper\/2017\/file\/3f5ee243547dee91fbd053c1c4a845aa-Paper.pdf"},{"key":"e_1_3_3_2_41_2","unstructured":"Jason Wei Xuezhi Wang Dale Schuurmans Maarten Bosma Brian Ichter Fei Xia Ed Chi Quoc Le and Denny Zhou. 2023. Chain-of-Thought Prompting Elicits Reasoning in Large Language Models. arxiv:https:\/\/arXiv.org\/abs\/2201.11903\u00a0[cs.CL] https:\/\/arxiv.org\/abs\/2201.11903"},{"key":"e_1_3_3_2_42_2","doi-asserted-by":"publisher","unstructured":"Guangxuan Xiao Yuandong Tian Beidi Chen Song Han and Mike Lewis. 2023. Efficient Streaming Language Models with Attention Sinks. arXiv arXiv:2309.17453 (2023). 10.48550\/arXiv.2309.17453 arxiv:https:\/\/arXiv.org\/abs\/2309.17453\u00a0[cs.CL]","DOI":"10.48550\/arXiv.2309.17453"},{"key":"e_1_3_3_2_43_2","doi-asserted-by":"publisher","unstructured":"Xiao Xiong Zhaorui Chen Yue Liang Minghao Tian Jiaxing Shang Jiang Zhong and Dajiang Liu. 2025. DynaX: Sparse Attention Acceleration with Dynamic X:M Fine\u2011Grained Structured Pruning. Proceedings of the 30th ACM International Conference on Architectural Support for Programming Languages and Operating Systems (ASPLOS) 2 (2025) 260\u2013274. 10.1145\/3676641.3715991","DOI":"10.1145\/3676641.3715991"},{"key":"e_1_3_3_2_44_2","unstructured":"Shunyu Yao Jeffrey Zhao Dian Yu Nan Du Izhak Shafran Karthik Narasimhan and Yuan Cao. 2023. ReAct: Synergizing Reasoning and Acting in Language Models. arxiv:https:\/\/arXiv.org\/abs\/2210.03629\u00a0[cs.CL] https:\/\/arxiv.org\/abs\/2210.03629"},{"key":"e_1_3_3_2_45_2","unstructured":"Jingyang Yuan Huazuo Gao Damai Dai Junyu Luo Liang Zhao Zhengyan Zhang Zhenda Xie YX Wei Lean Wang Zhiping Xiao et\u00a0al. 2025. Native sparse attention: Hardware-aligned and natively trainable sparse attention. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2502.11089 (2025)."},{"key":"e_1_3_3_2_46_2","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA59077.2024.00066"}],"event":{"name":"MICRO 2025: 58th IEEE\/ACM International Symposium on Microarchitecture","location":"Seoul Korea","acronym":"MICRO 2025","sponsor":["SIGMICRO ACM Special Interest Group on Microarchitectural Research and Processing"]},"container-title":["Proceedings of the 58th IEEE\/ACM International Symposium on Microarchitecture"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3725843.3756062","content-type":"application\/pdf","content-version":"vor","intended-application":"syndication"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3725843.3756062","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,1,26]],"date-time":"2026-01-26T21:43:46Z","timestamp":1769463826000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3725843.3756062"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,17]]},"references-count":45,"alternative-id":["10.1145\/3725843.3756062","10.1145\/3725843"],"URL":"https:\/\/doi.org\/10.1145\/3725843.3756062","relation":{},"subject":[],"published":{"date-parts":[[2025,10,17]]},"assertion":[{"value":"2025-10-17","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}