{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,5]],"date-time":"2026-06-05T15:56:26Z","timestamp":1780674986339,"version":"3.54.1"},"publisher-location":"New York, NY, USA","reference-count":38,"publisher":"ACM","funder":[{"name":"The IITP-ITRC grant","award":["RS-2021-II21205"],"award-info":[{"award-number":["RS-2021-II21205"]}]},{"name":"The Institute of Information & Communications Technology Planning & Evaluation(IITP) grant","award":["00228970, II221170"],"award-info":[{"award-number":["00228970, II221170"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,6,21]]},"DOI":"10.1145\/3695053.3730990","type":"proceedings-article","created":{"date-parts":[[2025,6,20]],"date-time":"2025-06-20T16:43:11Z","timestamp":1750437791000},"page":"1746-1759","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":4,"title":["Avalanche: Optimizing Cache Utilization via Matrix Reordering for Sparse Matrix Multiplication Accelerator"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-9029-5898","authenticated-orcid":false,"given":"Gwangeun","family":"Byeon","sequence":"first","affiliation":[{"name":"Sungkyunkwan University, Suwon, Republic of Korea"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-6350-9015","authenticated-orcid":false,"given":"Seongwook","family":"Kim","sequence":"additional","affiliation":[{"name":"Sungkyunkwan University, Suwon, Republic of Korea"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-1504-1247","authenticated-orcid":false,"given":"Hyungjin","family":"Kim","sequence":"additional","affiliation":[{"name":"Sungkyunkwan University, Suwon, Republic of Korea"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-5862-5334","authenticated-orcid":false,"given":"Sukhyun","family":"Han","sequence":"additional","affiliation":[{"name":"Sungkyunkwan University, Suwon, Republic of Korea"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1744-1393","authenticated-orcid":false,"given":"Jinkwon","family":"Kim","sequence":"additional","affiliation":[{"name":"KAIST, Daejeon, Republic of Korea"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1732-4314","authenticated-orcid":false,"given":"Prashant","family":"Nair","sequence":"additional","affiliation":[{"name":"University of British Columbia, Vancouver, Republic of Korea"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5972-4789","authenticated-orcid":false,"given":"Taewook","family":"Kang","sequence":"additional","affiliation":[{"name":"Sungkyunkwan University, Suwon, Republic of Korea"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7842-125X","authenticated-orcid":false,"given":"Seokin","family":"Hong","sequence":"additional","affiliation":[{"name":"Sungkyunkwan University, Suwon, Republic of Korea"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,6,20]]},"reference":[{"key":"e_1_3_3_1_2_2","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA47549.2020.00029"},{"key":"e_1_3_3_1_3_2","doi-asserted-by":"publisher","DOI":"10.1109\/PACT52795.2021.00016"},{"key":"e_1_3_3_1_4_2","doi-asserted-by":"publisher","DOI":"10.1109\/ISPASS57527.2023.00029"},{"key":"e_1_3_3_1_5_2","doi-asserted-by":"crossref","unstructured":"Rajeev Balasubramonian Andrew\u00a0B Kahng Naveen Muralimanohar Ali Shafiee and Vaishnav Srinivas. 2017. CACTI 7: New tools for interconnect exploration in innovative off-chip memories. ACM Transactions on Architecture and Code Optimization (TACO) 14 2 (2017) 1\u201325.","DOI":"10.1145\/3085572"},{"key":"e_1_3_3_1_6_2","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS.2017.93"},{"key":"e_1_3_3_1_7_2","doi-asserted-by":"crossref","unstructured":"Jeff Bolz Ian Farmer Eitan Grinspun and Peter Schr\u00f6der. 2003. Sparse matrix solvers on the GPU: conjugate gradients and multigrid. ACM transactions on graphics (TOG) 22 3 (2003) 917\u2013924.","DOI":"10.1145\/882262.882364"},{"key":"e_1_3_3_1_8_2","doi-asserted-by":"publisher","DOI":"10.1145\/2925426.2926278"},{"key":"e_1_3_3_1_9_2","doi-asserted-by":"publisher","DOI":"10.1109\/PACT58117.2023.00041"},{"key":"e_1_3_3_1_10_2","doi-asserted-by":"crossref","unstructured":"Timothy\u00a0A Davis and Yifan Hu. 2011. The University of Florida sparse matrix collection. ACM Transactions on Mathematical Software (TOMS) 38 1 (2011) 1\u201325.","DOI":"10.1145\/2049662.2049663"},{"key":"e_1_3_3_1_11_2","doi-asserted-by":"crossref","unstructured":"Timothy\u00a0A Davis and Ekanathan Palamadai\u00a0Natarajan. 2010. Algorithm 907: KLU a direct sparse solver for circuit simulation problems. ACM Transactions on Mathematical Software (TOMS) 37 3 (2010) 1\u201317.","DOI":"10.1145\/1824801.1824814"},{"key":"e_1_3_3_1_12_2","doi-asserted-by":"publisher","DOI":"10.1109\/FCCM.2006.45"},{"key":"e_1_3_3_1_13_2","unstructured":"John\u00a0R Gilbert Steven Reinhardt and Viral\u00a0B Shah. 2006. High-performance graph algorithms from parallel sparse matrices. International Journal of High Performance Computing Applications 19 4 (2006) 495\u2013509."},{"key":"e_1_3_3_1_14_2","doi-asserted-by":"crossref","unstructured":"John\u00a0R Gilbert Steve Reinhardt and Viral\u00a0B Shah. 2008. A unified framework for numerical and combinatorial computing. Computing in Science & Engineering 10 2 (2008) 20\u201325.","DOI":"10.1109\/MCSE.2008.45"},{"key":"e_1_3_3_1_15_2","doi-asserted-by":"crossref","unstructured":"Song Han Xingyu Liu Huizi Mao Jing Pu Ardavan Pedram Mark\u00a0A Horowitz and William\u00a0J Dally. 2016. EIE: Efficient inference engine on compressed deep neural network. ACM SIGARCH Computer Architecture News 44 3 (2016) 243\u2013254.","DOI":"10.1145\/3007787.3001163"},{"key":"e_1_3_3_1_16_2","volume-title":"International Conference on Learning Representations (ICLR)","author":"Han Song","year":"2016","unstructured":"Song Han, Huizi Mao, and William\u00a0J Dally. 2016. Deep compression: Compressing deep neural networks with pruning, trained quantization and huffman coding. In International Conference on Learning Representations (ICLR)."},{"key":"e_1_3_3_1_17_2","unstructured":"Song Han Jeff Pool John Tran and William Dally. 2015. Learning both weights and connections for efficient neural network. Advances in neural information processing systems 28 (2015)."},{"key":"e_1_3_3_1_18_2","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA51647.2021.00017"},{"key":"e_1_3_3_1_19_2","doi-asserted-by":"publisher","DOI":"10.1145\/3613424.3623790"},{"key":"e_1_3_3_1_20_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICCD58817.2023.00070"},{"key":"e_1_3_3_1_21_2","doi-asserted-by":"crossref","unstructured":"Hyoukjun Kwon Ananda Samajdar and Tushar Krishna. 2018. Maeri: Enabling flexible dataflow mapping over dnn accelerators via reconfigurable interconnects. ACM SIGPLAN Notices 53 2 (2018) 461\u2013475.","DOI":"10.1145\/3296957.3173176"},{"key":"e_1_3_3_1_22_2","doi-asserted-by":"publisher","DOI":"10.1145\/3575693.3575706"},{"key":"e_1_3_3_1_23_2","doi-asserted-by":"crossref","unstructured":"Wai-Hung Liu and Andrew\u00a0H Sherman. 1976. Comparative analysis of the Cuthill\u2013McKee and the reverse Cuthill\u2013McKee ordering algorithms for sparse matrices. SIAM J. Numer. Anal. 13 2 (1976) 198\u2013213.","DOI":"10.1137\/0713020"},{"key":"e_1_3_3_1_24_2","doi-asserted-by":"publisher","DOI":"10.1145\/3620666.3651381"},{"key":"e_1_3_3_1_25_2","doi-asserted-by":"publisher","DOI":"10.1145\/3582016.3582069"},{"key":"e_1_3_3_1_26_2","volume-title":"GPU Technology Conference","volume":"12","author":"Naumov Maxim","year":"2010","unstructured":"Maxim Naumov, L Chien, Philippe Vandermersch, and Ujval Kapasi. 2010. Cusparse library. In GPU Technology Conference, Vol.\u00a012."},{"key":"e_1_3_3_1_27_2","doi-asserted-by":"crossref","unstructured":"Michael\u00a0K Ng and Zhaochen Zhu. 2019. Sparse matrix computation for air quality forecast data assimilation. Numerical Algorithms 80 (2019) 687\u2013707.","DOI":"10.1007\/s11075-018-0502-6"},{"key":"e_1_3_3_1_28_2","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA.2018.00067"},{"key":"e_1_3_3_1_29_2","doi-asserted-by":"crossref","unstructured":"Angshuman Parashar Minsoo Rhu Anurag Mukkara Antonio Puglielli Rangharajan Venkatesan Brucek Khailany Joel Emer Stephen\u00a0W Keckler and William\u00a0J Dally. 2017. SCNN: An accelerator for compressed-sparse convolutional neural networks. ACM SIGARCH computer architecture news 45 2 (2017) 27\u201340.","DOI":"10.1145\/3140659.3080254"},{"key":"e_1_3_3_1_30_2","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA47549.2020.00015"},{"key":"e_1_3_3_1_31_2","doi-asserted-by":"publisher","DOI":"10.1109\/MICRO50266.2020.00068"},{"key":"e_1_3_3_1_32_2","doi-asserted-by":"crossref","unstructured":"Manuel Then Moritz Kaufmann Fernando Chirigati Tuan-Anh Hoang-Vu Kien Pham Alfons Kemper Thomas Neumann and Huy\u00a0T Vo. 2014. The more the merrier: Efficient multi-source graph traversal. Proceedings of the VLDB Endowment 8 4 (2014) 449\u2013460.","DOI":"10.14778\/2735496.2735507"},{"key":"e_1_3_3_1_33_2","doi-asserted-by":"publisher","DOI":"10.1145\/3581784.3607046"},{"key":"e_1_3_3_1_34_2","doi-asserted-by":"crossref","unstructured":"Endong Wang Qing Zhang Bo Shen Guangyong Zhang Xiaowei Lu Qing Wu and Yajuan Wang. 2014. Intel math kernel library. High-Performance Computing on the Intel\u00ae Xeon Phi\u2122: How to Fully Exploit MIC Architectures (2014) 167\u2013188.","DOI":"10.1007\/978-3-319-06486-4_7"},{"key":"e_1_3_3_1_35_2","doi-asserted-by":"crossref","unstructured":"AN Yzelman and Rob\u00a0H Bisseling. 2009. Cache-oblivious sparse matrix\u2013vector multiplication by using sparse matrix partitioning methods. SIAM Journal on Scientific Computing 31 4 (2009) 3128\u20133154.","DOI":"10.1137\/080733243"},{"key":"e_1_3_3_1_36_2","doi-asserted-by":"publisher","DOI":"10.1145\/3445814.3446702"},{"key":"e_1_3_3_1_37_2","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA47549.2020.00030"},{"key":"e_1_3_3_1_38_2","doi-asserted-by":"publisher","DOI":"10.1109\/MICRO.2018.00011"},{"key":"e_1_3_3_1_39_2","doi-asserted-by":"publisher","DOI":"10.1145\/3352460.3358269"}],"event":{"name":"ISCA '25: Proceedings of the 52nd Annual International Symposium on Computer Architecture","location":"Tokyo Japan","acronym":"SIGARCH '25","sponsor":["SIGARCH ACM Special Interest Group on Computer Architecture"]},"container-title":["Proceedings of the 52nd Annual International Symposium on Computer Architecture"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3695053.3730990","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,21]],"date-time":"2025-06-21T10:57:47Z","timestamp":1750503467000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3695053.3730990"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,6,20]]},"references-count":38,"alternative-id":["10.1145\/3695053.3730990","10.1145\/3695053"],"URL":"https:\/\/doi.org\/10.1145\/3695053.3730990","relation":{},"subject":[],"published":{"date-parts":[[2025,6,20]]},"assertion":[{"value":"2025-06-20","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}