{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,6]],"date-time":"2026-05-06T15:58:32Z","timestamp":1778083112546,"version":"3.51.4"},"publisher-location":"New York, NY, USA","reference-count":35,"publisher":"ACM","funder":[{"name":"Institute of Information & communications Technology Planning & Evaluation","award":["RS-2024-00398745"],"award-info":[{"award-number":["RS-2024-00398745"]}]},{"name":"Institute of Information & communications Technology Planning & Evaluation","award":["RS-2024-00454666"],"award-info":[{"award-number":["RS-2024-00454666"]}]},{"DOI":"10.13039\/501100003725","name":"National Research Foundation of Korea","doi-asserted-by":"publisher","award":["RS-2023-00251287"],"award-info":[{"award-number":["RS-2023-00251287"]}],"id":[{"id":"10.13039\/501100003725","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,6,13]]},"DOI":"10.1145\/3735452.3735523","type":"proceedings-article","created":{"date-parts":[[2025,6,13]],"date-time":"2025-06-13T15:11:16Z","timestamp":1749827476000},"page":"3-15","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":1,"title":["SPARQ: An Accelerator Architecture for Large Language Models with Joint Sparsity and Quantization Techniques"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0003-3191-610X","authenticated-orcid":false,"given":"Seonggyu","family":"Choi","sequence":"first","affiliation":[{"name":"Sungkyunkwan University, Suwon, Republic of Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8705-7066","authenticated-orcid":false,"given":"Hyungmin","family":"Cho","sequence":"additional","affiliation":[{"name":"Sungkyunkwan University, Suwon, Republic of Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,6,13]]},"reference":[{"key":"e_1_3_2_2_1_1","doi-asserted-by":"publisher","DOI":"10.1145\/3085572"},{"key":"e_1_3_2_2_2_1","doi-asserted-by":"publisher","DOI":"10.5281\/zenodo.15385238"},{"key":"e_1_3_2_2_3_1","unstructured":"Tim Dettmers Mike Lewis Younes Belkada and Luke Zettlemoyer. 2022. LLM.int8(): 8-bit Matrix Multiplication for Transformers at Scale. In Advances in Neural Information Processing Systems (NeurIPS)."},{"key":"e_1_3_2_2_4_1","volume-title":"Proceedings of the 40th International Conference on Machine Learning (ICML).","author":"Dettmers Tim","year":"2023","unstructured":"Tim Dettmers and Luke Zettlemoyer. 2023. The case for 4-bit precision: k-bit inference scaling laws. In Proceedings of the 40th International Conference on Machine Learning (ICML)."},{"key":"e_1_3_2_2_5_1","doi-asserted-by":"publisher","DOI":"10.1109\/TVLSI.2022.3197282"},{"key":"e_1_3_2_2_6_1","volume-title":"Proceedings of the 40th International Conference on Machine Learning (ICML).","author":"Frantar Elias","year":"2023","unstructured":"Elias Frantar and Dan Alistarh. 2023. SparseGPT: Massive Language Models Can Be Accurately Pruned in One-Shot. In Proceedings of the 40th International Conference on Machine Learning (ICML)."},{"key":"e_1_3_2_2_7_1","unstructured":"Elias Frantar and Dan Alistarh. 2024. Marlin: a fast 4-bit inference kernel for medium batchsizes. https:\/\/github.com\/IST-DASLab\/marlin"},{"key":"e_1_3_2_2_8_1","volume-title":"GPTQ: Accurate Post-training Compression for Generative Pretrained Transformers. In The Eleventh International Conference on Learning Representations (ICLR).","author":"Frantar Elias","year":"2023","unstructured":"Elias Frantar, Saleh Ashkboos, Torsten Hoefler, and Dan Alistarh. 2023. GPTQ: Accurate Post-training Compression for Generative Pretrained Transformers. In The Eleventh International Conference on Learning Representations (ICLR)."},{"key":"e_1_3_2_2_9_1","volume-title":"Forty-first International Conference on Machine Learning (ICML).","author":"Guo Jinyang","year":"2024","unstructured":"Jinyang Guo, Jianyu Wu, Zining Wang, Jiaheng Liu, Ge Yang, Yifu Ding, Ruihao Gong, Haotong Qin, and Xianglong Liu. 2024. Compressing Large Language Models by Joint Sparsification and Quantization. In Forty-first International Conference on Machine Learning (ICML)."},{"key":"e_1_3_2_2_10_1","volume-title":"FIGNA: Integer Unit-Based Accelerator Design for FP-INT GEMM Preserving Numerical Accuracy. In 2024 IEEE International Symposium on High-Performance Computer Architecture (HPCA).","author":"Jang Jaeyong","year":"2024","unstructured":"Jaeyong Jang, Yulhwa Kim, Juheun Lee, and Jae-Joon Kim. 2024. FIGNA: Integer Unit-Based Accelerator Design for FP-INT GEMM Preserving Numerical Accuracy. In 2024 IEEE International Symposium on High-Performance Computer Architecture (HPCA)."},{"key":"e_1_3_2_2_11_1","volume-title":"SDQ: Sparse Decomposed Quantization for LLM Inference. ArXiv, arXiv:2406.13868","author":"Jeong Geonhwa","year":"2024","unstructured":"Geonhwa Jeong, Po-An Tsai, Stephen W. Keckler, and Tushar Krishna. 2024. SDQ: Sparse Decomposed Quantization for LLM Inference. ArXiv, arXiv:2406.13868"},{"key":"e_1_3_2_2_12_1","unstructured":"Albert Q. Jiang Alexandre Sablayrolles Arthur Mensch Chris Bamford Devendra Singh Chaplot Diego de las Casas Florian Bressand Gianna Lengyel Guillaume Lample Lucile Saulnier L\u00e9lio Renard Lavaud Marie-Anne Lachaux Pierre Stock Teven Le Scao Thibaut Lavril Thomas Wang Timoth\u00e9e Lacroix and William El Sayed. 2023. Mistral 7B. arXiv preprint arXiv:2310.06825"},{"key":"e_1_3_2_2_13_1","volume-title":"Industrial Product. In 2021 ACM\/IEEE 48th Annual International Symposium on Computer Architecture (ISCA).","author":"Jouppi Norman P.","year":"2021","unstructured":"Norman P. Jouppi, Doe Hyun Yoon, Matthew Ashcraft, Mark Gottscho, Thomas B. Jablin, George Kurian, James Laudon, Sheng Li, Peter Ma, Xiaoyu Ma, Thomas Norrie, Nishant Patil, Sushma Prasad, Cliff Young, Zongwei Zhou, and David Patterson. 2021. Ten Lessons From Three Generations Shaped Google\u2019s TPUv4i : Industrial Product. In 2021 ACM\/IEEE 48th Annual International Symposium on Computer Architecture (ISCA)."},{"key":"e_1_3_2_2_14_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCD58817.2023.00070"},{"key":"e_1_3_2_2_15_1","volume-title":"Winning Both the Accuracy of Floating Point Activation and the Simplicity of Integer Arithmetic. In The Eleventh International Conference on Learning Representations (ICLR).","author":"Kim Yulhwa","year":"2023","unstructured":"Yulhwa Kim, Jaeyong Jang, Jehun Lee, Jihoon Park, Jeonghoon Kim, Byeongwook Kim, Baeseong Park, Se Jung Kwon, Dongsoo Lee, and Jae-Joon Kim. 2023. Winning Both the Accuracy of Floating Point Activation and the Simplicity of Integer Arithmetic. In The Eleventh International Conference on Learning Representations (ICLR)."},{"key":"e_1_3_2_2_16_1","unstructured":"Teven Le Scao Angela Fan Christopher Akiki Ellie Pavlick Suzana Ili\u0107 Daniel Hesslow Roman Castagn\u00e9 Alexandra Sasha Luccioni Fran\u00e7ois Yvon Matthias Gall\u00e9 Jonathan Tow Alexander M. Rush Stella Biderman Albert Webson Pawan Sasanka Ammanamanchi Thomas Wang Beno\u00eet Sagot Niklas Muennighoff Albert Villanova del Moral Olatunji Ruwase Rachel Bawden Stas Bekman Angelina McMillan-Major Iz Beltagy Huu Nguyen Lucile Saulnier Samson Tan Pedro Ortiz Suarez Victor Sanh Hugo Lauren\u00e7on Yacine Jernite Julien Launay Margaret Mitchell and Colin Raffel. 2022. BLOOM: A 176B-Parameter Open-Access Multilingual Language Model. arXiv preprint arXiv:2211.05100"},{"key":"e_1_3_2_2_17_1","unstructured":"Yun Li Lin Niu Xipeng Zhang Kai Liu Jianchen Zhu and Zhanhui Kang. 2024. E-Sparse: Boosting the Large Language Model Inference through Entropy-based N:M Sparsity. arXiv preprint arXiv:2310.15929"},{"key":"e_1_3_2_2_18_1","volume-title":"Proceedings of Machine Learning and Systems (MLSys).","author":"Lin Ji","year":"2024","unstructured":"Ji Lin, Jiaming Tang, Haotian Tang, Shang Yang, Wei-Ming Chen, Wei-Chen Wang, Guangxuan Xiao, Xingyu Dang, Chuang Gan, and Song Han. 2024. AWQ: Activation-aware Weight Quantization for On-Device LLM Compression and Acceleration. In Proceedings of Machine Learning and Systems (MLSys)."},{"key":"e_1_3_2_2_19_1","doi-asserted-by":"publisher","DOI":"10.1109\/LCA.2020.2979965"},{"key":"e_1_3_2_2_20_1","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA53966.2022.00049"},{"key":"e_1_3_2_2_21_1","volume-title":"Jeff Pool, Darko Stosic, Dusan Stosic, Ganesh Venkatesh, Chong Yu, and Paulius Micikevicius.","author":"Mishra Asit","year":"2021","unstructured":"Asit Mishra, Jorge Albericio Latorre, Jeff Pool, Darko Stosic, Dusan Stosic, Ganesh Venkatesh, Chong Yu, and Paulius Micikevicius. 2021. Accelerating Sparse Deep Neural Networks. arXiv preprint, arXiv:2104.08378"},{"key":"e_1_3_2_2_22_1","volume-title":"Fine-Grained DRAM: Energy-Efficient DRAM for Extreme Bandwidth Systems. In 2017 50th Annual IEEE\/ACM International Symposium on Microarchitecture (MICRO).","author":"O\u2019Connor Mike","unstructured":"Mike O\u2019Connor, Niladrish Chatterjee, Donghyuk Lee, John Wilson, Aditya Agrawal, Stephen W. Keckler, and William J. Dally. 2017. Fine-Grained DRAM: Energy-Efficient DRAM for Extreme Bandwidth Systems. In 2017 50th Annual IEEE\/ACM International Symposium on Microarchitecture (MICRO)."},{"key":"e_1_3_2_2_23_1","volume-title":"The Twelfth International Conference on Learning Representations (ICLR).","author":"Park Gunho","year":"2024","unstructured":"Gunho Park, Baeseong park, Minsub Kim, Sungjae Lee, Jeonghoon Kim, Beomseok Kwon, Se Jung Kwon, Byeongwook Kim, Youngjoo Lee, and Dongsoo Lee. 2024. LUT-GEMM: Quantized Matrix Multiplication based on LUTs for Efficient Inference in Large-Scale Generative Language Models. In The Twelfth International Conference on Learning Representations (ICLR)."},{"key":"e_1_3_2_2_24_1","volume-title":"Movement Pruning: Adaptive Sparsity by Fine-Tuning. In Advances in Neural Information Processing Systems (NeurIPS).","author":"Sanh Victor","year":"2020","unstructured":"Victor Sanh, Thomas Wolf, and Alexander Rush. 2020. Movement Pruning: Adaptive Sparsity by Fine-Tuning. In Advances in Neural Information Processing Systems (NeurIPS)."},{"key":"e_1_3_2_2_25_1","unstructured":"Wenqi Shao Mengzhao Chen Zhaoyang Zhang Peng Xu Lirui Zhao Zhiqian Li Kaipeng Zhang Peng Gao Yu Qiao and Ping Luo. 2024. OmniQuant: Omnidirectionally Calibrated Quantization for Large Language Models. arXiv preprint arXiv:2308.13137"},{"key":"e_1_3_2_2_26_1","volume-title":"The Twelfth International Conference on Learning Representations (ICLR).","author":"Sun Mingjie","year":"2024","unstructured":"Mingjie Sun, Zhuang Liu, Anna Bair, and J Zico Kolter. 2024. A Simple and Effective Pruning Approach for Large Language Models. In The Twelfth International Conference on Learning Representations (ICLR)."},{"key":"e_1_3_2_2_27_1","unstructured":"Llama Team. 2024. The Llama 3 Herd of Models. arXiv preprint arXiv:2407.21783"},{"key":"e_1_3_2_2_28_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.emnlp-main.496"},{"key":"e_1_3_2_2_29_1","volume-title":"Outlier Suppression: Pushing the Limit of Low-bit Transformer Language Models. In Advances in Neural Information Processing Systems (NeurIPS).","author":"Wei Xiuying","year":"2022","unstructured":"Xiuying Wei, Yunchen Zhang, Xiangguo Zhang, Ruihao Gong, Shanghang Zhang, Qi Zhang, Fengwei Yu, and Xianglong Liu. 2022. Outlier Suppression: Pushing the Limit of Low-bit Transformer Language Models. In Advances in Neural Information Processing Systems (NeurIPS)."},{"key":"e_1_3_2_2_30_1","volume-title":"Proceedings of the 40th International Conference on Machine Learning (ICML).","author":"Wu Xiaoxia","year":"2023","unstructured":"Xiaoxia Wu, Cheng Li, Reza Yazdani Aminabadi, Zhewei Yao, and Yuxiong He. 2023. Understanding INT4 quantization for language models: latency speedup, composability, and failure cases. In Proceedings of the 40th International Conference on Machine Learning (ICML)."},{"key":"e_1_3_2_2_31_1","volume-title":"Proceedings of the 40th International Conference on Machine Learning (ICML).","author":"Xiao Guangxuan","year":"2023","unstructured":"Guangxuan Xiao, Ji Lin, Mickael Seznec, Hao Wu, Julien Demouth, and Song Han. 2023. SmoothQuant: Accurate and Efficient Post-Training Quantization for Large Language Models. In Proceedings of the 40th International Conference on Machine Learning (ICML)."},{"key":"e_1_3_2_2_32_1","first-page":"1440","article-title":"S2 Engine","volume":"71","author":"Yang Jianlei","year":"2022","unstructured":"Jianlei Yang, Wenzhi Fu, Xingzhou Cheng, Xucheng Ye, Pengcheng Dai, and Weisheng Zhao. 2022. S2 Engine: A Novel Systolic Architecture for Sparse Convolutional Neural Networks. IEEE Trans. Comput., 71, 6 (2022), 1440\u20131452.","journal-title":"IEEE Trans. Comput."},{"key":"e_1_3_2_2_33_1","volume-title":"Minjia Zhang, Xiaoxia Wu, Conglong Li, and Yuxiong He.","author":"Yao Zhewei","year":"2022","unstructured":"Zhewei Yao, Reza Yazdani Aminabadi, Minjia Zhang, Xiaoxia Wu, Conglong Li, and Yuxiong He. 2022. ZeroQuant: Efficient and Affordable Post-Training Quantization for Large-Scale Transformers. In Advances in Neural Information Processing Systems (NeurIPS)."},{"key":"e_1_3_2_2_34_1","doi-asserted-by":"publisher","DOI":"10.1145\/3649329.3656497"},{"key":"e_1_3_2_2_35_1","volume-title":"Todor Mihaylov, Myle Ott, Sam Shleifer, Kurt Shuster, Daniel Simig, Punit Singh Koura, Anjali Sridhar, Tianlu Wang, and Luke Zettlemoyer.","author":"Zhang Susan","year":"2022","unstructured":"Susan Zhang, Stephen Roller, Naman Goyal, Mikel Artetxe, Moya Chen, Shuohui Chen, Christopher Dewan, Mona Diab, Xian Li, Xi Victoria Lin, Todor Mihaylov, Myle Ott, Sam Shleifer, Kurt Shuster, Daniel Simig, Punit Singh Koura, Anjali Sridhar, Tianlu Wang, and Luke Zettlemoyer. 2022. OPT: Open Pre-trained Transformer Language Models. arXiv preprint, arXiv:2205.01068"}],"event":{"name":"LCTES '25: 26th ACM SIGPLAN\/SIGBED International Conference on Languages, Compilers, and Tools for Embedded Systems","location":"Seoul Republic of Korea","acronym":"LCTES '25","sponsor":["SIGPLAN ACM Special Interest Group on Programming Languages","SIGBED ACM Special Interest Group on Embedded Systems"]},"container-title":["Proceedings of the 26th ACM SIGPLAN\/SIGBED International Conference on Languages, Compilers, and Tools for Embedded Systems"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3735452.3735523","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,7,16]],"date-time":"2025-07-16T07:12:58Z","timestamp":1752649978000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3735452.3735523"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,6,13]]},"references-count":35,"alternative-id":["10.1145\/3735452.3735523","10.1145\/3735452"],"URL":"https:\/\/doi.org\/10.1145\/3735452.3735523","relation":{},"subject":[],"published":{"date-parts":[[2025,6,13]]},"assertion":[{"value":"2025-06-13","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}