{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,22]],"date-time":"2026-07-22T11:04:38Z","timestamp":1784718278938,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":14,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,7,26]],"date-time":"2026-07-26T00:00:00Z","timestamp":1785024000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,7,26]]},"DOI":"10.1145\/3785462.3815902","type":"proceedings-article","created":{"date-parts":[[2026,7,22]],"date-time":"2026-07-22T10:41:43Z","timestamp":1784716903000},"page":"1-5","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Phase-Wise Analysis of LLM Inference Acceleration on GPU, CPU, and Edge Device"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-5272-5105","authenticated-orcid":false,"given":"Subhransu","family":"Das","sequence":"first","affiliation":[{"name":"The Ohio State University, Columbus, OH, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-1170-8571","authenticated-orcid":false,"given":"Jiaming","family":"Cheng","sequence":"additional","affiliation":[{"name":"The Ohio State University, Columbus, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9094-1722","authenticated-orcid":false,"given":"Swathi","family":"Vallabhajosyula","sequence":"additional","affiliation":[{"name":"The Ohio State University, Columbus, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0362-5087","authenticated-orcid":false,"given":"Brijesh","family":"Soni","sequence":"additional","affiliation":[{"name":"The Ohio State University, Columbus, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0093-8560","authenticated-orcid":false,"given":"Rajiv","family":"Ramnath","sequence":"additional","affiliation":[{"name":"The Ohio State University, Columbus, OH, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,7,26]]},"reference":[{"key":"e_1_3_3_2_2_2","volume-title":"International Conference on Learning Representations (ICLR)","author":"Dao Tri","year":"2024","unstructured":"Tri Dao. 2024. FlashAttention-2: Faster Attention with Better Parallelism and Work Partitioning. In International Conference on Learning Representations (ICLR)."},{"key":"e_1_3_3_2_3_2","doi-asserted-by":"crossref","unstructured":"Tri Dao Dan Fu Stefano Ermon Atri Rudra and Christopher R\u00e9. 2022. Flashattention: Fast and memory-efficient exact attention with io-awareness. Advances in neural information processing systems 35 (2022) 16344\u201316359.","DOI":"10.52202\/068431-1189"},{"key":"e_1_3_3_2_4_2","unstructured":"Dao-AILab. 2026. Flash Attention Issue #2381: Qwen3.5 crashes with illegal memory access due to 3D position_ids misinterpreted as packed sequence. https:\/\/github.com\/Dao-AILab\/flash-attention\/issues\/2381."},{"key":"e_1_3_3_2_5_2","doi-asserted-by":"publisher","DOI":"10.52202\/075280-0441"},{"key":"e_1_3_3_2_6_2","first-page":"4171","volume-title":"Proceedings of the 2019 conference of the North American chapter of the association for computational linguistics: human language technologies, volume 1 (long and short papers)","author":"Devlin Jacob","year":"2019","unstructured":"Jacob Devlin, Ming-Wei Chang, Kenton Lee, and Kristina Toutanova. 2019. Bert: Pre-training of deep bidirectional transformers for language understanding. In Proceedings of the 2019 conference of the North American chapter of the association for computational linguistics: human language technologies, volume 1 (long and short papers). 4171\u20134186."},{"key":"e_1_3_3_2_7_2","unstructured":"Georgi Gerganov and contributors. 2023. llama.cpp. https:\/\/github.com\/ggml-org\/llama.cpp."},{"key":"e_1_3_3_2_8_2","doi-asserted-by":"publisher","DOI":"10.1145\/3600006.3613165"},{"key":"e_1_3_3_2_9_2","first-page":"19274","volume-title":"International Conference on Machine Learning","author":"Leviathan Yaniv","year":"2023","unstructured":"Yaniv Leviathan, Matan Kalman, and Yossi Matias. 2023. Fast inference from transformers via speculative decoding. In International Conference on Machine Learning. PMLR, 19274\u201319286."},{"key":"e_1_3_3_2_10_2","unstructured":"Jinhao Li Jiaming Xu Shan Huang Yonghua Chen Wen Li Jun Liu Yaoxiu Lian Jiayi Pan Li Ding Hao Zhou et\u00a0al. 2024. Large language model inference acceleration: A comprehensive hardware perspective. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2410.04466 (2024)."},{"key":"e_1_3_3_2_11_2","doi-asserted-by":"crossref","unstructured":"Humza Naveed Asad\u00a0Ullah Khan Shi Qiu Muhammad Saqib Saeed Anwar Muhammad Usman Naveed Akhtar Nick Barnes and Ajmal Mian. 2025. A comprehensive overview of large language models. ACM Transactions on Intelligent Systems and Technology 16 5 (2025) 1\u201372.","DOI":"10.1145\/3744746"},{"key":"e_1_3_3_2_12_2","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA59077.2024.00019"},{"key":"e_1_3_3_2_13_2","volume-title":"Improving language understanding by generative pre-training","author":"Radford Alec","year":"2018","unstructured":"Alec Radford, Karthik Narasimhan, Tim Salimans, and Ilya Sutskever. 2018. Improving language understanding by generative pre-training. Technical Report. OpenAI."},{"key":"e_1_3_3_2_14_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D16-1264"},{"key":"e_1_3_3_2_15_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.acl-long.631"}],"event":{"name":"PEARC '26: Practice and Experience in Advanced Research Computing","location":"Minneapolis MN USA","acronym":"PEARC '26","sponsor":["SIGHPC ACM Special Interest Group on High Performance Computing, Special Interest Group on High Performance Computing","SIGAPP ACM Special Interest Group on Applied Computing"]},"container-title":["Proceedings of the Practice and Experience in Advanced Research Computing 2026: Resilient Roots + Empowered Communities"],"original-title":[],"deposited":{"date-parts":[[2026,7,22]],"date-time":"2026-07-22T10:42:26Z","timestamp":1784716946000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3785462.3815902"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,26]]},"references-count":14,"alternative-id":["10.1145\/3785462.3815902","10.1145\/3785462"],"URL":"https:\/\/doi.org\/10.1145\/3785462.3815902","relation":{},"subject":[],"published":{"date-parts":[[2026,7,26]]},"assertion":[{"value":"2026-07-26","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}