{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,23]],"date-time":"2026-04-23T07:55:34Z","timestamp":1776930934994,"version":"3.51.2"},"publisher-location":"New York, NY, USA","reference-count":52,"publisher":"ACM","license":[{"start":{"date-parts":[[2025,11,15]],"date-time":"2025-11-15T00:00:00Z","timestamp":1763164800000},"content-version":"vor","delay-in-days":0,"URL":"http:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"US Department of Energy","award":["P-1-06007"],"award-info":[{"award-number":["P-1-06007"]}]},{"name":"National Science Foundation of China","award":["92270206"],"award-info":[{"award-number":["92270206"]}]},{"DOI":"10.13039\/100006235","name":"Lawrence Berkeley National Laboratory","doi-asserted-by":"publisher","award":["DE-AC02-05CH11231"],"award-info":[{"award-number":["DE-AC02-05CH11231"]}],"id":[{"id":"10.13039\/100006235","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,11,16]]},"DOI":"10.1145\/3712285.3759792","type":"proceedings-article","created":{"date-parts":[[2025,11,12]],"date-time":"2025-11-12T16:05:39Z","timestamp":1762963539000},"page":"1477-1494","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Caracal: A GPU-Resident Sparse LU Solver with Lightweight Fine-Grained Scheduling"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-3332-2092","authenticated-orcid":false,"given":"Jie","family":"Ren","sequence":"first","affiliation":[{"name":"Computer Science, King Abdullah University of Science and Technology, Thuwal, Saudi Arabia"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-5086-4649","authenticated-orcid":false,"given":"Tingxuan","family":"Zhong","sequence":"additional","affiliation":[{"name":"Computer Science, King Abdullah University of Science and Technology, Thuwal, Saudi Arabia"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0741-6602","authenticated-orcid":false,"given":"Yuxi","family":"Hong","sequence":"additional","affiliation":[{"name":"Lawrence Berkeley National Laboratory, Berkeley, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-6107-2501","authenticated-orcid":false,"given":"Guofeng","family":"Feng","sequence":"additional","affiliation":[{"name":"Chinese Academy of Sciences, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-7753-6663","authenticated-orcid":false,"given":"Xincheng","family":"Wang","sequence":"additional","affiliation":[{"name":"Xidian University, Xi'an, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8539-8326","authenticated-orcid":false,"given":"Weile","family":"Jia","sequence":"additional","affiliation":[{"name":"Chinese Academy of Sciences, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6897-1095","authenticated-orcid":false,"given":"Hatem","family":"Ltaief","sequence":"additional","affiliation":[{"name":"Applied Mathematics and Computational Sciences, King Abdullah University of Science and Technology, Thuwal, Saudi Arabia"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4052-7224","authenticated-orcid":false,"given":"David Elliot","family":"Keyes","sequence":"additional","affiliation":[{"name":"Applied Mathematics and Computational Sciences, King Abdullah University of Science and Technology, Thuwal, Saudi Arabia"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,11,15]]},"reference":[{"key":"e_1_3_3_3_2_2","doi-asserted-by":"crossref","unstructured":"Patrick\u00a0R Amestoy Timothy\u00a0A Davis and Iain\u00a0S Duff. 1996. An approximate minimum degree ordering algorithm. SIAM J. Matrix Anal. Appl. (1996).","DOI":"10.1137\/S0895479894278952"},{"key":"e_1_3_3_3_3_2","doi-asserted-by":"crossref","unstructured":"Patrick\u00a0R Amestoy Timothy\u00a0A Davis and Iain\u00a0S Duff. 2004. Algorithm 837: AMD an approximate minimum degree ordering algorithm. ACM Trans. Math. Software (2004).","DOI":"10.1145\/1024074.1024081"},{"key":"e_1_3_3_3_4_2","volume-title":"International Workshop on Applied Parallel Computing","author":"Amestoy Patrick\u00a0R","year":"2000","unstructured":"Patrick\u00a0R Amestoy, Iain\u00a0S Duff, Jean-Yves L\u2019Excellent, and Jacko Koster. 2000. MUMPS: a general purpose distributed memory sparse solver. In International Workshop on Applied Parallel Computing. Springer."},{"key":"e_1_3_3_3_5_2","volume-title":"Computational fluid dynamics","author":"Anderson John\u00a0David","year":"1995","unstructured":"John\u00a0David Anderson and John Wendt. 1995. Computational fluid dynamics. Springer."},{"key":"e_1_3_3_3_6_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-33518-1_40"},{"key":"e_1_3_3_3_7_2","unstructured":"C\u00e9dric Augonnet Samuel Thibault and Raymond Namyst. 2010. StarPU: a runtime system for scheduling tasks over accelerator-based multicore machines. Ph.\u00a0D. Dissertation. INRIA."},{"key":"e_1_3_3_3_8_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-03869-3_80"},{"key":"e_1_3_3_3_9_2","doi-asserted-by":"publisher","DOI":"10.1017\/CBO9780511800955"},{"key":"e_1_3_3_3_10_2","unstructured":"Xiaoming Chen Ling Ren Yu Wang and Huazhong Yang. 2014. GPU-accelerated sparse LU factorization for circuit simulation with performance modeling. IEEE Transactions on Parallel and Distributed Systems (2014)."},{"key":"e_1_3_3_3_11_2","doi-asserted-by":"publisher","DOI":"10.1109\/ASPDAC.2012.6164974"},{"key":"e_1_3_3_3_12_2","doi-asserted-by":"crossref","unstructured":"Xiaoming Chen Yu Wang and Huazhong Yang. 2013. NICSLU: An adaptive sparse matrix solver for parallel circuit simulation. IEEE transactions on computer-aided design of integrated circuits and systems (2013).","DOI":"10.1109\/TCAD.2012.2217964"},{"key":"e_1_3_3_3_13_2","unstructured":"Ludovic Court\u00e8s. 2013. C language extensions for hybrid CPU\/GPU programming with StarPU. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1304.0878 (2013)."},{"key":"e_1_3_3_3_14_2","doi-asserted-by":"crossref","unstructured":"Timothy\u00a0A Davis. 2004. Algorithm 832: UMFPACK V4. 3\u2014an unsymmetric-pattern multifrontal method. ACM Trans. Math. Software (2004).","DOI":"10.1145\/992200.992206"},{"key":"e_1_3_3_3_15_2","doi-asserted-by":"publisher","DOI":"10.1137\/1.9780898718881"},{"key":"e_1_3_3_3_16_2","doi-asserted-by":"crossref","unstructured":"Timothy\u00a0A Davis John\u00a0R Gilbert Stefan\u00a0I Larimore and Esmond\u00a0G Ng. 2004. Algorithm 836: COLAMD a column approximate minimum degree ordering algorithm. ACM Trans. Math. Software (2004).","DOI":"10.1145\/1024074.1024080"},{"key":"e_1_3_3_3_17_2","doi-asserted-by":"crossref","unstructured":"Timothy\u00a0A Davis John\u00a0R Gilbert Stefan\u00a0I Larimore and Esmond\u00a0G Ng. 2004. A column approximate minimum degree ordering algorithm. ACM Trans. Math. Software (2004).","DOI":"10.1145\/1024074.1024079"},{"key":"e_1_3_3_3_18_2","doi-asserted-by":"crossref","unstructured":"Timothy\u00a0A Davis and Ekanathan Palamadai\u00a0Natarajan. 2010. Algorithm 907: KLU a direct sparse solver for circuit simulation problems. ACM Trans. Math. Software (2010).","DOI":"10.1145\/1824801.1824814"},{"key":"e_1_3_3_3_19_2","doi-asserted-by":"crossref","unstructured":"James\u00a0W Demmel John\u00a0R Gilbert and Xiaoye\u00a0S Li. 1999. An asynchronous parallel supernodal algorithm for sparse Gaussian elimination. SIAM J. Matrix Anal. Appl. (1999).","DOI":"10.1137\/S0895479897317685"},{"key":"e_1_3_3_3_20_2","doi-asserted-by":"publisher","DOI":"10.1093\/acprof:oso\/9780198508380.001.0001"},{"key":"e_1_3_3_3_21_2","doi-asserted-by":"crossref","unstructured":"Iain\u00a0S Duff and John\u00a0K Reid. 1983. The multifrontal solution of indefinite sparse symmetric linear. ACM Trans. Math. Software (1983).","DOI":"10.1145\/356044.356047"},{"key":"e_1_3_3_3_22_2","doi-asserted-by":"crossref","unstructured":"Stanley\u00a0C Eisenstat and Joseph\u00a0WH Liu. 1992. Exploiting structural symmetry in unsymmetric sparse symbolic factorization. SIAM J. Matrix Anal. Appl. (1992).","DOI":"10.1137\/0613017"},{"key":"e_1_3_3_3_23_2","doi-asserted-by":"crossref","unstructured":"Stanley\u00a0C Eisenstat and Joseph\u00a0WH Liu. 1993. Exploiting structural symmetry in a sparse partial pivoting code. SIAM Journal on Scientific Computing (1993).","DOI":"10.21236\/ADA264129"},{"key":"e_1_3_3_3_24_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-69583-4_13"},{"key":"e_1_3_3_3_25_2","doi-asserted-by":"publisher","DOI":"10.1145\/3581784.3607050"},{"key":"e_1_3_3_3_26_2","doi-asserted-by":"crossref","unstructured":"William\u00a0D Gropp and David\u00a0E Keyes. 1992. Domain decomposition methods in computational fluid dynamics. International Journal for Numerical Methods in Fluids (1992).","DOI":"10.1002\/fld.1650140203"},{"key":"e_1_3_3_3_27_2","doi-asserted-by":"crossref","unstructured":"Kai He Sheldon X-D Tan Hai Wang and Guoyong Shi. 2015. GPU-accelerated parallel sparse LU factorization method for fast circuit analysis. IEEE Transactions on Very Large Scale Integration (VLSI) Systems (2015).","DOI":"10.1109\/TVLSI.2015.2421287"},{"key":"e_1_3_3_3_28_2","doi-asserted-by":"crossref","unstructured":"Pascal H\u00e9non Pierre Ramet and Jean Roman. 2002. PaStiX: a high-performance parallel direct solver for sparse symmetric positive definite systems. Parallel Comput. (2002).","DOI":"10.1016\/S0167-8191(01)00141-7"},{"key":"e_1_3_3_3_29_2","volume-title":"Structural analysis","author":"Hibbeler Russell\u00a0Charles","year":"2006","unstructured":"Russell\u00a0Charles Hibbeler and Kiang-Hwee Tan. 2006. Structural analysis. Pearson Prentice Hall Upper Saddle River."},{"key":"e_1_3_3_3_30_2","doi-asserted-by":"publisher","DOI":"10.5555\/3571885.3571891"},{"key":"e_1_3_3_3_31_2","doi-asserted-by":"crossref","unstructured":"Pascal H\u00e9non Pierre Ramet and Jean Roman. 2000. PaStiX: A Parallel Sparse Direct Solver Based on a Static Scheduling for Mixed 1D\/2D Block Distributions. Springer Berlin Heidelberg (2000).","DOI":"10.1007\/3-540-45591-4_70"},{"key":"e_1_3_3_3_32_2","unstructured":"George Karypis and Vipin Kumar. 1997. METIS: A software package for partitioning unstructured graphs partitioning meshes and computing fill-reducing orderings of sparse matrices. (1997)."},{"key":"e_1_3_3_3_33_2","volume-title":"Structural analysis","author":"Kassimali Aslam","year":"1999","unstructured":"Aslam Kassimali and Amit Prashant. 1999. Structural analysis. PWS Pub."},{"key":"e_1_3_3_3_34_2","unstructured":"David\u00a0Elliot Keyes and MD Smooke. 1987. Flame sheet starting estimates for counterflow diffusion flame problems. J. Comput. Phys. (1987)."},{"key":"e_1_3_3_3_35_2","unstructured":"Wai-Kong Lee Ramachandra Achar and Michel\u00a0S Nakhla. 2018. Dynamic GPU parallel sparse LU factorization for fast circuit simulation. IEEE Transactions on Very Large Scale Integration (VLSI) Systems (2018)."},{"key":"e_1_3_3_3_36_2","unstructured":"Xiaoye\u00a0S Li. 2005. An overview of SuperLU: Algorithms implementation and user interface. ACM Trans. Math. Software (2005)."},{"key":"e_1_3_3_3_37_2","unstructured":"Xiaoye\u00a0S. Li and James\u00a0W. Demmel. 2003. SuperLU_DIST: A scalable distributed-memory sparse direct solver for unsymmetric linear systems. ACM Trans. Math. Software (2003)."},{"key":"e_1_3_3_3_38_2","volume-title":"Sparse Days 2023","author":"Lisito Alycia","year":"2023","unstructured":"Alycia Lisito, Mathieu Faverge, and Pierre Ramet. 2023. New parallel features in the sparse solver PaStiX. In Sparse Days 2023."},{"key":"e_1_3_3_3_39_2","doi-asserted-by":"publisher","DOI":"10.1109\/SC41406.2024.00012"},{"key":"e_1_3_3_3_40_2","volume-title":"Heat transfer","author":"Mills Anthony\u00a0F","year":"1992","unstructured":"Anthony\u00a0F Mills. 1992. Heat transfer. CRC Press."},{"key":"e_1_3_3_3_41_2","doi-asserted-by":"publisher","DOI":"10.1017\/CBO9780511841606"},{"key":"e_1_3_3_3_42_2","unstructured":"Dartzi Pan and Harvard Lomax. 1988. A new approximate LU factorization scheme for the Reynolds-averaged Navier-Stokes equations. AIAA journal (1988)."},{"key":"e_1_3_3_3_43_2","doi-asserted-by":"crossref","unstructured":"K\u00a0Guru Prasad JH Kane David\u00a0Elliot Keyes and C Balakrishna. 1994. Preconditioned Krylov solvers for BEA. Internat. J. Numer. Methods Engrg. (1994).","DOI":"10.1002\/nme.1620371003"},{"key":"e_1_3_3_3_44_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-19836-6_40"},{"key":"e_1_3_3_3_45_2","doi-asserted-by":"publisher","DOI":"10.23919\/ISC.2025.11017733"},{"key":"e_1_3_3_3_46_2","doi-asserted-by":"publisher","DOI":"10.1145\/2228360.2228565"},{"key":"e_1_3_3_3_47_2","doi-asserted-by":"publisher","DOI":"10.1145\/3330345.3330357"},{"key":"e_1_3_3_3_48_2","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS.2018.00100"},{"key":"e_1_3_3_3_49_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-09873-9_41"},{"key":"e_1_3_3_3_50_2","doi-asserted-by":"crossref","unstructured":"Olaf Schenk Klaus G\u00e4rtner Wolfgang Fichtner and Andreas Stricker. 2001. PARDISO: a high-performance serial and parallel sparse linear solver in semiconductor device simulation. Future Generation Computer Systems (2001).","DOI":"10.1016\/S0167-739X(00)00076-5"},{"key":"e_1_3_3_3_51_2","doi-asserted-by":"crossref","unstructured":"Alisa\u00a0V Trofimova Andr\u00e9s\u00a0E Tejada-Mart\u00ednez Kenneth\u00a0E Jansen and Richard\u00a0T Lahey\u00a0Jr. 2009. Direct numerical simulation of turbulent channel flows using a stabilized finite element method. Computers & Fluids (2009).","DOI":"10.1016\/j.compfluid.2008.10.003"},{"key":"e_1_3_3_3_52_2","doi-asserted-by":"publisher","DOI":"10.1109\/DAC56929.2023.10247767"},{"key":"e_1_3_3_3_53_2","doi-asserted-by":"publisher","DOI":"10.1109\/DAC18074.2021.9586141"}],"event":{"name":"SC '25: The International Conference for High Performance Computing, Networking, Storage and Analysis","location":"St. Louis MO USA","acronym":"SC '25","sponsor":["SIGHPC ACM Special Interest Group on High Performance Computing, Special Interest Group on High Performance Computing"]},"container-title":["Proceedings of the International Conference for High Performance Computing, Networking, Storage and Analysis"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/abs\/10.1145\/3712285.3759792","content-type":"text\/html","content-version":"vor","intended-application":"syndication"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3712285.3759792","content-type":"application\/pdf","content-version":"vor","intended-application":"syndication"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3712285.3759792","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,3,11]],"date-time":"2026-03-11T18:53:34Z","timestamp":1773255214000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3712285.3759792"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,11,15]]},"references-count":52,"alternative-id":["10.1145\/3712285.3759792","10.1145\/3712285"],"URL":"https:\/\/doi.org\/10.1145\/3712285.3759792","relation":{},"subject":[],"published":{"date-parts":[[2025,11,15]]},"assertion":[{"value":"2025-11-15","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}