{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,5]],"date-time":"2026-03-05T15:40:06Z","timestamp":1772725206205,"version":"3.50.1"},"publisher-location":"New York, NY, USA","reference-count":67,"publisher":"ACM","license":[{"start":{"date-parts":[[2023,10,28]],"date-time":"2023-10-28T00:00:00Z","timestamp":1698451200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-sa\/4.0\/"}],"funder":[{"DOI":"10.13039\/100000001","name":"National Science Foundation","doi-asserted-by":"publisher","award":["CCF-2217099"],"award-info":[{"award-number":["CCF-2217099"]}],"id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Wistron Corporation"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2023,10,28]]},"DOI":"10.1145\/3613424.3623783","type":"proceedings-article","created":{"date-parts":[[2023,12,8]],"date-time":"2023-12-08T17:22:15Z","timestamp":1702056135000},"page":"91-104","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":9,"title":["Spatula: A Hardware Accelerator for Sparse Matrix Factorization"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-8797-7476","authenticated-orcid":false,"given":"Axel","family":"Feldmann","sequence":"first","affiliation":[{"name":"Massachusetts Institute of Technology, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2453-2904","authenticated-orcid":false,"given":"Daniel","family":"Sanchez","sequence":"additional","affiliation":[{"name":"Massachusetts Institute of Technology, United States of America"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2023,12,8]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.5555\/3571885.3571919"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"crossref","unstructured":"Patrick\u00a0R Amestoy Iain\u00a0S Duff Jean-Yves L\u2019Excellent and Jacko Koster. 2001. MUMPS: a general purpose distributed memory sparse solver. In Applied Parallel Computing.  Patrick\u00a0R Amestoy Iain\u00a0S Duff Jean-Yves L\u2019Excellent and Jacko Koster. 2001. MUMPS: a general purpose distributed memory sparse solver. In Applied Parallel Computing.","DOI":"10.1007\/3-540-70734-4_16"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA47549.2020.00029"},{"key":"e_1_3_2_1_4_1","volume-title":"Alex Woo, and Maurice Yarrow.","author":"Bailey David","year":"1995","unstructured":"David Bailey , Tim Harris , William Saphir , Rob Van Der\u00a0Wijngaart , Alex Woo, and Maurice Yarrow. 1995 . The NAS parallel benchmarks 2.0. Technical Report NAS-95-020. NASA Ames Research Center . David Bailey, Tim Harris, William Saphir, Rob Van Der\u00a0Wijngaart, Alex Woo, and Maurice Yarrow. 1995. The NAS parallel benchmarks 2.0. Technical Report NAS-95-020. NASA Ames Research Center."},{"key":"e_1_3_2_1_5_1","volume-title":"Covariance prediction via convex optimization. Optimization and Engineering","author":"Barratt Shane","year":"2022","unstructured":"Shane Barratt and Stephen Boyd . 2022. Covariance prediction via convex optimization. Optimization and Engineering ( 2022 ). Shane Barratt and Stephen Boyd. 2022. Covariance prediction via convex optimization. Optimization and Engineering (2022)."},{"key":"e_1_3_2_1_7_1","article-title":"NICSLU: An adaptive sparse matrix solver for parallel circuit simulation","volume":"32","author":"Chen Xiaoming","year":"2013","unstructured":"Xiaoming Chen , Yu Wang , and Huazhong Yang . 2013 . NICSLU: An adaptive sparse matrix solver for parallel circuit simulation . IEEE Transactions on Computer-Aided Design of Integrated Circuits and Systems (IEEE TCAD) 32 , 2 (2013). Xiaoming Chen, Yu Wang, and Huazhong Yang. 2013. NICSLU: An adaptive sparse matrix solver for parallel circuit simulation. IEEE Transactions on Computer-Aided Design of Integrated Circuits and Systems (IEEE TCAD) 32, 2 (2013).","journal-title":"IEEE Transactions on Computer-Aided Design of Integrated Circuits and Systems (IEEE TCAD)"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1145\/1391989.1391995"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1145\/3276493"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1145\/3503222.3507706"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1109\/ISSCC19947.2020.9062947"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/ISSCC19947.2020.9062947"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1145\/992200.992206"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"crossref","unstructured":"Timothy\u00a0A Davis. 2006. Direct methods for sparse linear systems. SIAM.  Timothy\u00a0A Davis. 2006. Direct methods for sparse linear systems. SIAM.","DOI":"10.1137\/1.9780898718881"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1145\/2491491.2491498"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1145\/2049662.2049663"},{"key":"e_1_3_2_1_17_1","volume-title":"Scientific Computing in Electrical Engineering (SCEE","author":"Davis A","year":"2010","unstructured":"Timothy\u00a0 A Davis and E\u00a0Palamadai Natarajan . 2011. Sparse matrix methods for circuit simulation problems . In Scientific Computing in Electrical Engineering (SCEE 2010 ). Timothy\u00a0A Davis and E\u00a0Palamadai Natarajan. 2011. Sparse matrix methods for circuit simulation problems. In Scientific Computing in Electrical Engineering (SCEE 2010)."},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1145\/1824801.1824814"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1145\/356044.356047"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"crossref","unstructured":"Wei Ge Mengnan Zhao Cheng Wu and Jun He. 2011. The Design and Implementation of DDR PHY Static Low-Power Optimization Strategies. In Communication Systems and Information Technology.  Wei Ge Mengnan Zhao Cheng Wu and Jun He. 2011. The Design and Implementation of DDR PHY Static Low-Power Optimization Strategies. In Communication Systems and Information Technology.","DOI":"10.1007\/978-3-642-21762-3_1"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS.2011.44"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"crossref","unstructured":"Pieter Ghysels and Ryan Synk. 2022. High performance sparse multifrontal solvers on modern GPUs. Parallel Comput. (2022).  Pieter Ghysels and Ryan Synk. 2022. High performance sparse multifrontal solvers on modern GPUs. Parallel Comput. (2022).","DOI":"10.1016\/j.parco.2022.102897"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1137\/0909058"},{"key":"e_1_3_2_1_24_1","volume-title":"Matrix computations","author":"Golub H","unstructured":"Gene\u00a0 H Golub and Charles\u00a0 F Van\u00a0Loan . 2013. Matrix computations . JHU press . Gene\u00a0H Golub and Charles\u00a0F Van\u00a0Loan. 2013. Matrix computations. JHU press."},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA.2011.5749755"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1145\/2155620.2155623"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1145\/3038228.3038237"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1145\/3392717.3392751"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1145\/3352460.3358275"},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1109\/FPT.2009.5377665"},{"key":"e_1_3_2_1_31_1","volume-title":"A scalable sparse Cholesky based approach for learning high-dimensional covariance matrices in ordered data. Machine Learning 108, 12","author":"Khare Kshitij","year":"2019","unstructured":"Kshitij Khare , Sang-Yun Oh , Syed Rahman , and Bala Rajaratnam . 2019. A scalable sparse Cholesky based approach for learning high-dimensional covariance matrices in ordered data. Machine Learning 108, 12 ( 2019 ). Kshitij Khare, Sang-Yun Oh, Syed Rahman, and Bala Rajaratnam. 2019. A scalable sparse Cholesky based approach for learning high-dimensional covariance matrices in ordered data. Machine Learning 108, 12 (2019)."},{"key":"e_1_3_2_1_32_1","volume-title":"Sparse matrix factorization in the implicit finite element method on petascale architecture. Computer Methods in Applied Mechanics and Engineering","author":"Koric Seid","year":"2016","unstructured":"Seid Koric and Anshul Gupta . 2016. Sparse matrix factorization in the implicit finite element method on petascale architecture. Computer Methods in Applied Mechanics and Engineering ( 2016 ). Seid Koric and Anshul Gupta. 2016. Sparse matrix factorization in the implicit finite element method on petascale architecture. Computer Methods in Applied Mechanics and Engineering (2016)."},{"key":"e_1_3_2_1_33_1","volume-title":"Evaluation of massively parallel linear sparse solvers on unstructured finite element meshes. Computers & Structures","author":"Koric Seid","year":"2014","unstructured":"Seid Koric , Qiyue Lu , and Erman Guleryuz . 2014. Evaluation of massively parallel linear sparse solvers on unstructured finite element meshes. Computers & Structures ( 2014 ). Seid Koric, Qiyue Lu, and Erman Guleryuz. 2014. Evaluation of massively parallel linear sparse solvers on unstructured finite element meshes. Computers & Structures (2014)."},{"key":"e_1_3_2_1_34_1","volume-title":"Sparse Matrix Proceedings.","author":"Kung Hsiang\u00a0Tsung","year":"1979","unstructured":"Hsiang\u00a0Tsung Kung and Charles\u00a0 E Leiserson . 1979 . Systolic arrays (for VLSI) . In Sparse Matrix Proceedings. Hsiang\u00a0Tsung Kung and Charles\u00a0E Leiserson. 1979. Systolic arrays (for VLSI). In Sparse Matrix Proceedings."},{"key":"e_1_3_2_1_35_1","unstructured":"Jean-Yves L\u2019Excellent. 2012. Multifrontal methods: parallelism memory usage and numerical aspects. Ph.\u00a0D. Dissertation. Ecole Normale Sup\u00e9rieure de Lyon.  Jean-Yves L\u2019Excellent. 2012. Multifrontal methods: parallelism memory usage and numerical aspects. Ph.\u00a0D. Dissertation. Ecole Normale Sup\u00e9rieure de Lyon."},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1145\/1089014.1089017"},{"key":"e_1_3_2_1_37_1","volume-title":"Proceedings of the Ninth SIAM Conference on Parallel Processing for Scientific Computing (PPSC).","author":"Li S","year":"1999","unstructured":"Xiaoye\u00a0 S Li and James Demmel . 1999 . A Scalable Sparse Direct Solver Using Static Pivoting .. In Proceedings of the Ninth SIAM Conference on Parallel Processing for Scientific Computing (PPSC). Xiaoye\u00a0S Li and James Demmel. 1999. A Scalable Sparse Direct Solver Using Static Pivoting.. In Proceedings of the Ninth SIAM Conference on Parallel Processing for Scientific Computing (PPSC)."},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1145\/3575693.3575706"},{"key":"e_1_3_2_1_39_1","series-title":"SIAM Rev. (1992)","volume-title":"The multifrontal method for sparse matrix solution: Theory and practice","author":"Liu WH","unstructured":"Joseph\u00a0 WH Liu . 1992. The multifrontal method for sparse matrix solution: Theory and practice . SIAM Rev. (1992) . Joseph\u00a0WH Liu. 1992. The multifrontal method for sparse matrix solution: Theory and practice. SIAM Rev. (1992)."},{"key":"e_1_3_2_1_41_1","volume-title":"Proceedings of the ACM SIGMETRICS International Conference on Measurement and Modeling of Computer Systems.","author":"Mauer J.","year":"2002","unstructured":"Carl\u00a0 J. Mauer , Mark\u00a0 D. Hill , and David\u00a0 A. Wood . 2002 . Full-system timing-first simulation . In Proceedings of the ACM SIGMETRICS International Conference on Measurement and Modeling of Computer Systems. Carl\u00a0J. Mauer, Mark\u00a0D. Hill, and David\u00a0A. Wood. 2002. Full-system timing-first simulation. In Proceedings of the ACM SIGMETRICS International Conference on Measurement and Modeling of Computer Systems."},{"key":"e_1_3_2_1_42_1","volume-title":"The Exascale Computing Project. Computing in Science & Engineering","author":"Messina Paul","year":"2017","unstructured":"Paul Messina . 2017. The Exascale Computing Project. Computing in Science & Engineering ( 2017 ). Paul Messina. 2017. The Exascale Computing Project. Computing in Science & Engineering (2017)."},{"key":"e_1_3_2_1_43_1","unstructured":"Micron. 2018. High Bandwidth Memory with ECC. https:\/\/media-www.micron.com\/-\/media\/client\/global\/documents\/products\/data-sheet\/dram\/hbm2e\/8gb_and_16gb_hbm2e_dram.pdf.  Micron. 2018. High Bandwidth Memory with ECC. https:\/\/media-www.micron.com\/-\/media\/client\/global\/documents\/products\/data-sheet\/dram\/hbm2e\/8gb_and_16gb_hbm2e_dram.pdf."},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1145\/1168857.1168878"},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1145\/3582016.3582069"},{"key":"e_1_3_2_1_46_1","volume-title":"Parallel sparse matrix solution for circuit simulation on FPGAs","author":"Nechma Tarek","year":"2014","unstructured":"Tarek Nechma and Mark Zwolinski . 2014. Parallel sparse matrix solution for circuit simulation on FPGAs . IEEE Trans. Comput . ( 2014 ). Tarek Nechma and Mark Zwolinski. 2014. Parallel sparse matrix solution for circuit simulation on FPGAs. IEEE Trans. Comput. (2014)."},{"key":"e_1_3_2_1_47_1","unstructured":"NVIDIA. 2017. NVIDIA Tesla V100 GPU Architecture. https:\/\/images.nvidia.com\/content\/volta-architecture\/pdf\/volta-architecture-whitepaper.pdf.  NVIDIA. 2017. NVIDIA Tesla V100 GPU Architecture. https:\/\/images.nvidia.com\/content\/volta-architecture\/pdf\/volta-architecture-whitepaper.pdf."},{"key":"e_1_3_2_1_48_1","unstructured":"NVIDIA. 2020. NVIDIA DGX Station A100 System Architecture. https:\/\/images.nvidia.com\/aem-dam\/Solutions\/Data-Center\/nvidia-dgx-station-a100-system-architecture-white-paper.pdf.  NVIDIA. 2020. NVIDIA DGX Station A100 System Architecture. https:\/\/images.nvidia.com\/aem-dam\/Solutions\/Data-Center\/nvidia-dgx-station-a100-system-architecture-white-paper.pdf."},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.1145\/3466752.3480134"},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA.2018.00067"},{"key":"e_1_3_2_1_51_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCAD.2011.2176730"},{"key":"e_1_3_2_1_52_1","unstructured":"Rambus Inc.2020. White paper: HBM2E and GDDR6: Memory Solutions for AI.  Rambus Inc.2020. White paper: HBM2E and GDDR6: Memory Solutions for AI."},{"key":"e_1_3_2_1_53_1","doi-asserted-by":"publisher","DOI":"10.1145\/3503222.3507705"},{"key":"e_1_3_2_1_54_1","doi-asserted-by":"crossref","unstructured":"Steven\u00a0C Rennich Darko Stosic and Timothy\u00a0A Davis. 2016. Accelerating sparse Cholesky factorization on GPUs. Parallel Comput. (2016).  Steven\u00a0C Rennich Darko Stosic and Timothy\u00a0A Davis. 2016. Accelerating sparse Cholesky factorization on GPUs. Parallel Comput. (2016).","DOI":"10.1016\/j.parco.2016.06.004"},{"key":"e_1_3_2_1_55_1","doi-asserted-by":"publisher","DOI":"10.1145\/3352460.3358330"},{"key":"e_1_3_2_1_56_1","doi-asserted-by":"publisher","DOI":"10.1145\/356004.356006"},{"key":"e_1_3_2_1_57_1","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA47549.2020.00062"},{"key":"e_1_3_2_1_58_1","volume-title":"Proceedings of the 20th International Meshing Roundtable.","author":"Staten L","year":"2012","unstructured":"Matthew\u00a0 L Staten , Steven\u00a0 J Owen , Suzanne\u00a0 M Shontz , Andrew\u00a0 G Salinger , and Todd\u00a0 S Coffey . 2012 . A comparison of mesh morphing methods for 3D shape optimization . In Proceedings of the 20th International Meshing Roundtable. Matthew\u00a0L Staten, Steven\u00a0J Owen, Suzanne\u00a0M Shontz, Andrew\u00a0G Salinger, and Todd\u00a0S Coffey. 2012. A comparison of mesh morphing methods for 3D shape optimization. In Proceedings of the 20th International Meshing Roundtable."},{"key":"e_1_3_2_1_59_1","doi-asserted-by":"publisher","DOI":"10.2514\/6.2007-4322"},{"key":"e_1_3_2_1_60_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-01766-7"},{"key":"e_1_3_2_1_61_1","volume-title":"Proceedings of the","author":"Thornton E","year":"1964","unstructured":"James\u00a0 E Thornton . 1964 . Parallel operation in the Control Data 6600 . In Proceedings of the October 27-29, 1964, Fall Joint Computer Conference, Part II: Very High Speed Computer Systems. James\u00a0E Thornton. 1964. Parallel operation in the Control Data 6600. In Proceedings of the October 27-29, 1964, Fall Joint Computer Conference, Part II: Very High Speed Computer Systems."},{"key":"e_1_3_2_1_62_1","doi-asserted-by":"publisher","DOI":"10.1145\/1736020.1736044"},{"key":"e_1_3_2_1_63_1","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA52012.2021.00088"},{"key":"e_1_3_2_1_64_1","volume-title":"Learning structured sparsity in deep neural networks. Advances in neural information processing systems (NeurIPS)","author":"Wen Wei","year":"2016","unstructured":"Wei Wen , Chunpeng Wu , Yandan Wang , Yiran Chen , and Hai Li. 2016. Learning structured sparsity in deep neural networks. Advances in neural information processing systems (NeurIPS) ( 2016 ). Wei Wen, Chunpeng Wu, Yandan Wang, Yiran Chen, and Hai Li. 2016. Learning structured sparsity in deep neural networks. Advances in neural information processing systems (NeurIPS) (2016)."},{"key":"e_1_3_2_1_65_1","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA47549.2020.00063"},{"key":"e_1_3_2_1_66_1","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA51647.2021.00055"},{"key":"e_1_3_2_1_67_1","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA56546.2023.10071080"},{"key":"e_1_3_2_1_68_1","doi-asserted-by":"publisher","DOI":"10.1145\/3445814.3446702"},{"key":"e_1_3_2_1_69_1","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA47549.2020.00030"}],"event":{"name":"MICRO '23: 56th Annual IEEE\/ACM International Symposium on Microarchitecture","location":"Toronto ON Canada","acronym":"MICRO '23","sponsor":["SIGMICRO ACM Special Interest Group on Microarchitectural Research and Processing"]},"container-title":["56th Annual IEEE\/ACM International Symposium on Microarchitecture"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3613424.3623783","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3613424.3623783","content-type":"application\/pdf","content-version":"vor","intended-application":"syndication"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3613424.3623783","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T16:36:30Z","timestamp":1750178190000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3613424.3623783"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,10,28]]},"references-count":67,"alternative-id":["10.1145\/3613424.3623783","10.1145\/3613424"],"URL":"https:\/\/doi.org\/10.1145\/3613424.3623783","relation":{},"subject":[],"published":{"date-parts":[[2023,10,28]]},"assertion":[{"value":"2023-12-08","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}