{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,2]],"date-time":"2026-07-02T23:48:36Z","timestamp":1783036116786,"version":"3.54.6"},"publisher-location":"New York, NY, USA","reference-count":111,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,1,28]]},"DOI":"10.1145\/3774934.3786442","type":"proceedings-article","created":{"date-parts":[[2026,1,28]],"date-time":"2026-01-28T15:25:57Z","timestamp":1769613957000},"page":"369-383","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":1,"title":["Trojan Horse: Aggregate-and-Batch for Scaling Up Sparse Direct Solvers on GPU Clusters"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-0527-5209","authenticated-orcid":false,"given":"Yida","family":"Li","sequence":"first","affiliation":[{"name":"China University of Petroleum-Beijing, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-3869-5453","authenticated-orcid":false,"given":"Siwei","family":"Zhang","sequence":"additional","affiliation":[{"name":"China University of Petroleum-Beijing, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-6203-5361","authenticated-orcid":false,"given":"Yiduo","family":"Niu","sequence":"additional","affiliation":[{"name":"China University of Petroleum-Beijing, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-2799-2201","authenticated-orcid":false,"given":"Yang","family":"Du","sequence":"additional","affiliation":[{"name":"China University of Petroleum-Beijing, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2927-362X","authenticated-orcid":false,"given":"Qingxiao","family":"Sun","sequence":"additional","affiliation":[{"name":"China University of Petroleum-Beijing, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0632-9494","authenticated-orcid":false,"given":"Zhou","family":"Jin","sequence":"additional","affiliation":[{"name":"China University of Petroleum-Beijing, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2150-5759","authenticated-orcid":false,"given":"Weifeng","family":"Liu","sequence":"additional","affiliation":[{"name":"China University of Petroleum-Beijing, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,1,28]]},"reference":[{"key":"e_1_3_2_2_1_1","volume-title":"Linear Algebra Software for Large-Scale Accelerated Multicore Computing. Acta Numerica, 25","author":"Abdelfattah A.","year":"2016","unstructured":"A. Abdelfattah, H. Anzt, J. Dongarra, M. Gates, A. Haidar, J. Kurzak, P. Luszczek, S. Tomov, I. Yamazaki, and A. YarKhan. 2016. Linear Algebra Software for Large-Scale Accelerated Multicore Computing. Acta Numerica, 25 (2016)."},{"key":"e_1_3_2_2_2_1","volume-title":"Matrix Multiplication on Batches of Small Matrices in Half and Half-Complex Precisions. J. Parallel and Distrib. Comput., 145","author":"Abdelfattah Ahmad","year":"2020","unstructured":"Ahmad Abdelfattah, Stanimire Tomov, and Jack Dongarra. 2020. Matrix Multiplication on Batches of Small Matrices in Half and Half-Complex Precisions. J. Parallel and Distrib. Comput., 145 (2020)."},{"key":"e_1_3_2_2_3_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPDS.2021.3084071"},{"key":"e_1_3_2_2_4_1","doi-asserted-by":"publisher","DOI":"10.1109\/71.395403"},{"key":"e_1_3_2_2_5_1","doi-asserted-by":"publisher","DOI":"10.1137\/130938505"},{"key":"e_1_3_2_2_6_1","doi-asserted-by":"publisher","DOI":"10.1137\/23M1568600"},{"key":"e_1_3_2_2_7_1","doi-asserted-by":"publisher","DOI":"10.1145\/1024074.1024081"},{"key":"e_1_3_2_2_8_1","doi-asserted-by":"publisher","DOI":"10.1016\/S0045-7825(99)00242-X"},{"key":"e_1_3_2_2_9_1","doi-asserted-by":"publisher","DOI":"10.1137\/S0895479899358194"},{"key":"e_1_3_2_2_10_1","volume-title":"MUMPS: A General Purpose Distributed Memory Sparse Solver. In PARA.","author":"Amestoy Patrick R.","year":"2001","unstructured":"Patrick R. Amestoy, Iain S. Duff, Jean-Yves L\u2019Excellent, and Jacko Koster. 2001. MUMPS: A General Purpose Distributed Memory Sparse Solver. In PARA."},{"key":"e_1_3_2_2_11_1","doi-asserted-by":"publisher","DOI":"10.1137\/S0895479802419877"},{"key":"e_1_3_2_2_12_1","doi-asserted-by":"publisher","DOI":"10.1137\/17M1151882"},{"key":"e_1_3_2_2_13_1","volume-title":"Quintana-Ort\u00ed","author":"Anzt Hartwig","year":"2017","unstructured":"Hartwig Anzt, Jack. Dongarra, Goran Flegar, and Enrique S. Quintana-Ort\u00ed. 2017. Variable-Size Batched LU for Small Matrices and Its Integration into Block-Jacobi Preconditioning. In ICPP."},{"key":"e_1_3_2_2_14_1","volume-title":"Li","author":"Anzt Hartwig","year":"2024","unstructured":"Hartwig Anzt, Axel Huebl, and Xiaoye S. Li. 2024. Then and Now: Improving Software Portability, Productivity, and 100\u00d7 Performance. Computing in Science & Engineering, 26, 1 (2024)."},{"key":"e_1_3_2_2_15_1","doi-asserted-by":"crossref","unstructured":"C\u00e9dric Augonnet Samuel Thibault Raymond Namyst and Pierre-Andr\u00e9 Wacrenier. 2009. StarPU: A Unified Platform for Task Scheduling on Heterogeneous Multicore Architectures. In Euro-Par.","DOI":"10.1007\/978-3-642-03869-3_80"},{"key":"e_1_3_2_2_16_1","volume-title":"xSDK Foundations: Toward an Extreme-scale Scientific Software Development Kit. Supercomputing Frontiers and Innovations, 4, 1","author":"Bartlett Roscoe","year":"2017","unstructured":"Roscoe Bartlett, Irina Demeshko, Todd Gamblin, Glenn Hammond, Michael Heroux, Jeffrey Johnson, Alicia Klinvex, Xiaoye Li, Lois McInnes, J. David Moulton, Daniel Osei-Kuffuor, Jason Sarich, Barry Smith, James Willenbring, and Ulrike Meier Yang. 2017. xSDK Foundations: Toward an Extreme-scale Scientific Software Development Kit. Supercomputing Frontiers and Innovations, 4, 1 (2017)."},{"key":"e_1_3_2_2_17_1","volume-title":"Manolis Papadakis, Galen Shipman, Patrick McCormick, Michael Garland, and Alex Aiken.","author":"Bauer Michael","year":"2021","unstructured":"Michael Bauer, Wonchan Lee, Elliott Slaughter, Zhihao Jia, Mario Di Renzo, Manolis Papadakis, Galen Shipman, Patrick McCormick, Michael Garland, and Alex Aiken. 2021. Scaling implicit parallelism via dynamic control replication. In PPoPP."},{"key":"e_1_3_2_2_18_1","doi-asserted-by":"crossref","unstructured":"Michael Bauer Elliott Slaughter Sean Treichler Wonchan Lee Michael Garland and Alex Aiken. 2023. Visibility Algorithms for Dynamic Dependence Analysis and Distributed Coherence. In PPoPP.","DOI":"10.1145\/3572848.3577515"},{"key":"e_1_3_2_2_19_1","volume-title":"CUSP: Generic Parallel Algorithms for Sparse Matrix and Graph Computations.","author":"Bell Nathan","year":"2009","unstructured":"Nathan Bell and Michael Garland. 2009. CUSP: Generic Parallel Algorithms for Sparse Matrix and Graph Computations."},{"key":"e_1_3_2_2_20_1","doi-asserted-by":"crossref","unstructured":"Noel Chalmers Jakub Kurzak Damon Mcdougall and Paul Bauman. 2023. Optimizing High-Performance Linpack for Exascale Accelerated Architectures. In SC.","DOI":"10.1145\/3581784.3607066"},{"key":"e_1_3_2_2_21_1","doi-asserted-by":"publisher","DOI":"10.1145\/3267101"},{"key":"e_1_3_2_2_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCAD.2012.2217964"},{"key":"e_1_3_2_2_23_1","doi-asserted-by":"publisher","DOI":"10.1145\/1391989.1391995"},{"key":"e_1_3_2_2_24_1","doi-asserted-by":"crossref","unstructured":"Helin Cheng Wenxuan Li Yuechen Lu and Weifeng Liu. 2023. HASpGEMM: Heterogeneity-Aware Sparse General Matrix-Matrix Multiplication on Modern Asymmetric Multicore Processors. In ICPP. isbn:9798400708435","DOI":"10.1145\/3605573.3605611"},{"key":"e_1_3_2_2_25_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPDS.2019.2938172"},{"key":"e_1_3_2_2_26_1","doi-asserted-by":"publisher","DOI":"10.1177\/10943420241288567"},{"key":"e_1_3_2_2_27_1","doi-asserted-by":"crossref","unstructured":"Michel Cosnard and Laura Grigori. 2001. A parallel algorithm for sparse symbolic LU factorization without pivoting on out-of-core matrices. In ICS.","DOI":"10.1145\/377792.377823"},{"key":"e_1_3_2_2_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPDS.2014.2333526"},{"key":"e_1_3_2_2_29_1","unstructured":"Roshan Dathathri Gurbinder Gill Loc Hoang Vishwesh Jatala Keshav Pingali V. Krishna Nandivada Hoang-Vu Dang and Marc Snir. 2024. Gluon-Async: A Bulk-Asynchronous System for Distributed and Heterogeneous Graph Analytics. In PACT."},{"key":"e_1_3_2_2_30_1","doi-asserted-by":"publisher","DOI":"10.1145\/992200.992206"},{"key":"e_1_3_2_2_31_1","doi-asserted-by":"publisher","DOI":"10.5555\/1196434"},{"key":"e_1_3_2_2_32_1","doi-asserted-by":"publisher","DOI":"10.1145\/2049662.2049670"},{"key":"e_1_3_2_2_33_1","doi-asserted-by":"publisher","DOI":"10.1145\/2049662.2049663"},{"key":"e_1_3_2_2_34_1","doi-asserted-by":"publisher","DOI":"10.1145\/1824801.1824814"},{"key":"e_1_3_2_2_35_1","doi-asserted-by":"publisher","DOI":"10.1137\/S0895479895291765"},{"key":"e_1_3_2_2_36_1","doi-asserted-by":"publisher","DOI":"10.1137\/S0895479897317685"},{"key":"e_1_3_2_2_37_1","article-title":"A set of level 3 basic linear algebra subprograms","volume":"16","author":"Croz Jeremy Du","year":"1990","unstructured":"Jack. Dongarra, Jeremy Du Croz, Sven Hammarling, and I. S. Duff. 1990. A set of level 3 basic linear algebra subprograms. ACM Trans. Math. Software, 16, 1 (1990).","journal-title":"ACM Trans. Math. Software"},{"key":"e_1_3_2_2_38_1","article-title":"PLASMA: Parallel Linear Algebra Software for Multicore Using OpenMP","volume":"45","author":"Gates Mark","year":"2019","unstructured":"Jack. Dongarra, Mark Gates, Azzam Haidar, Jakub Kurzak, Piotr Luszczek, Panruo Wu, Ichitaro Yamazaki, Asim Yarkhan, Maksims Abalenkovs, Negin Bagherpour, Sven Hammarling, Jakub \u0160\u00edstek, David Stevens, Mawussi Zounon, and Samuel D. Relton. 2019. PLASMA: Parallel Linear Algebra Software for Multicore Using OpenMP. ACM Trans. Math. Software, 45, 2 (2019).","journal-title":"ACM Trans. Math. Software"},{"key":"e_1_3_2_2_39_1","volume-title":"Proc. IEEE, 106","author":"Gates Mark","year":"2018","unstructured":"Jack. Dongarra, Mark Gates, Jakub Kurzak, Piotr Luszczek, and Yaohung M. Tsai. 2018. Autotuning Numerical Dense Linear Algebra for Batched Computation With GPU Hardware Accelerators. Proc. IEEE, 106, 11 (2018)."},{"key":"e_1_3_2_2_40_1","volume-title":"Parallel implementation of multifrontal schemes. Parallel Comput., 3, 3","author":"Duff Iain S.","year":"1986","unstructured":"Iain S. Duff. 1986. Parallel implementation of multifrontal schemes. Parallel Comput., 3, 3 (1986)."},{"key":"e_1_3_2_2_41_1","volume-title":"Reid","author":"Duff Iain S.","year":"2017","unstructured":"Iain S. Duff, Albert M. Erisman, and John K. Reid. 2017. Direct Methods for Sparse Matrices. Oxford University Press."},{"key":"e_1_3_2_2_42_1","doi-asserted-by":"publisher","DOI":"10.1145\/356044.356047"},{"key":"e_1_3_2_2_43_1","volume-title":"Porting hypre to heterogeneous computer architectures: Strategies and experiences. Parallel Comput., 108","author":"Falgout Robert D.","year":"2021","unstructured":"Robert D. Falgout, Ruipeng Li, Bj\u00f6rn Sj\u00f6green, Lu Wang, and Ulrike Meier Yang. 2021. Porting hypre to heterogeneous computer architectures: Strategies and experiences. Parallel Comput., 108 (2021)."},{"key":"e_1_3_2_2_44_1","doi-asserted-by":"crossref","unstructured":"Xu Fu Bingbin Zhang Tengcheng Wang Wenhao Li Yuechen Lu Enxin Yi Jianqi Zhao Xiaohan Geng Fangying Li Jingwen Zhang Zhou Jin and Weifeng Liu. 2023. PanguLU: A Scalable Regular Two-Dimensional Block-Cyclic Sparse Direct Solver on Distributed Heterogeneous Systems. In SC.","DOI":"10.1145\/3581784.3607050"},{"key":"e_1_3_2_2_45_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPDS.2021.3090316"},{"key":"e_1_3_2_2_46_1","doi-asserted-by":"publisher","DOI":"10.1137\/1.9780898717846"},{"key":"e_1_3_2_2_47_1","volume-title":"Anne Benoit, Mathieu Faverge, Loris Marchal, Gr\u00e9goire Pichon, and Pierre Ramet.","author":"Gou Changjiang","year":"2020","unstructured":"Changjiang Gou, Ali Al Zoobi, Anne Benoit, Mathieu Faverge, Loris Marchal, Gr\u00e9goire Pichon, and Pierre Ramet. 2020. Improving Mapping for Sparse Direct Solvers. In Euro-Par."},{"key":"e_1_3_2_2_48_1","doi-asserted-by":"publisher","DOI":"10.1137\/050629343"},{"key":"e_1_3_2_2_49_1","volume-title":"Li","author":"Grigori Laura","year":"2002","unstructured":"Laura Grigori and Xiaoye S. Li. 2002. A new scheduling algorithm for parallel sparse LU factorization with static pivoting. In SC. isbn:076951524X"},{"key":"e_1_3_2_2_50_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPDS.2017.2783929"},{"key":"e_1_3_2_2_51_1","first-page":"6","article-title":"On finding approximate supernodes for an efficient ILU(k) factorization","volume":"34","author":"H\u00e9non Pascal","year":"2002","unstructured":"Pascal H\u00e9non, Pierre Ramet, and Jean Roman. 2002. On finding approximate supernodes for an efficient ILU(k) factorization. Parallel Comput., 34, 6-8 (2002), 345\u2013362.","journal-title":"Parallel Comput."},{"key":"e_1_3_2_2_52_1","doi-asserted-by":"publisher","DOI":"10.1016\/S0167-8191(01)00141-7"},{"key":"e_1_3_2_2_53_1","article-title":"Batched sparse direct solver design and evaluation in SuperLU_DIST","volume":"38","author":"Heroux Michael","year":"2024","unstructured":"Michael Heroux, Wajih Boukaram, Yuxi Hong, Yang Liu, Tianyi Shi, and Xiaoye S Li. 2024. Batched sparse direct solver design and evaluation in SuperLU_DIST. The International Journal of High Performance Computing Applications, 38, 6 (2024).","journal-title":"The International Journal of High Performance Computing Applications"},{"key":"e_1_3_2_2_54_1","doi-asserted-by":"crossref","unstructured":"Shintaro Iwasaki and Kenjiro Taura. 2016. Autotuning of a Cut-Off for Task Parallel Programs. In MCSoC.","DOI":"10.1145\/2967938.2967968"},{"key":"e_1_3_2_2_55_1","doi-asserted-by":"crossref","unstructured":"Zhou Jin Wenhao Li Yinuo Bai Tengcheng Wang Yicheng Lu and Weifeng Liu. 2024. Machine Learning and GPU Accelerated Sparse Linear Solvers for Transistor-Level Circuit Simulation: A Perspective Survey (Invited Paper). In ASP-DAC.","DOI":"10.1109\/ASP-DAC58780.2024.10473846"},{"key":"e_1_3_2_2_56_1","unstructured":"Changyeon Jo Hyunik Kim Hexiang Geng and Bernhard Egger. 2020. RackMem: A Tailored Caching Layer for Rack Scale Computing. In PACT."},{"key":"e_1_3_2_2_57_1","volume-title":"Peyton","author":"Karsavuran M. Ozan","year":"2025","unstructured":"M. Ozan Karsavuran, Esmond G. Ng, and Barry W. Peyton. 2025. GPU Accelerated Sparse Cholesky Factorization. In SC."},{"key":"e_1_3_2_2_58_1","doi-asserted-by":"crossref","unstructured":"Aditya Kashi Pratik Nayak Dhruva Kulkarni Aaron Scheinberg Paul Lin and Hartwig Anzt. 2022. Batched sparse iterative solvers on GPU for the collision operator for fusion plasma simulations. In IPDPS.","DOI":"10.1109\/IPDPS53621.2022.00024"},{"key":"e_1_3_2_2_59_1","doi-asserted-by":"crossref","unstructured":"Enver Kayaaslan and Bora U\u00e7ar. 2014. Reducing elimination tree height for parallel LU factorization of sparse unsymmetric matrices. In HiPC.","DOI":"10.1109\/HiPC.2014.7116880"},{"key":"e_1_3_2_2_60_1","doi-asserted-by":"publisher","DOI":"10.1177\/10943420211020803"},{"key":"e_1_3_2_2_61_1","doi-asserted-by":"crossref","unstructured":"Milind Kulkarni Patrick Carribault Keshav Pingali Ganesh Ramanarayanan Bruce Walter Kavita Bala and L. Paul Chew. 2008. Scheduling strategies for optimistic parallel execution of irregular programs. In SPAA.","DOI":"10.1145\/1378533.1378575"},{"key":"e_1_3_2_2_62_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPDS.2015.2481890"},{"key":"e_1_3_2_2_63_1","doi-asserted-by":"crossref","unstructured":"Xavier Lacoste Mathieu Faverge George Bosilca Pierre Ramet and Samuel Thibault. 2014. Taking Advantage of Hybrid Systems for Sparse Direct Solvers via Task-Based Runtimes. In IPDPS.","DOI":"10.1109\/IPDPSW.2014.9"},{"key":"e_1_3_2_2_64_1","volume-title":"Nikolopoulos","author":"Li Dong","year":"2010","unstructured":"Dong Li, Bronis R de Supinski, Martin Schulz, Kirk Cameron, and Dimitrios S. Nikolopoulos. 2010. Hybrid MPI\/OpenMP power-aware computing. In IPDPS."},{"key":"e_1_3_2_2_65_1","volume-title":"GPU-accelerated preconditioned iterative linear solvers. The Journal of Supercomputing, 63","author":"Li Ruipeng","year":"2013","unstructured":"Ruipeng Li and Yousef Saad. 2013. GPU-accelerated preconditioned iterative linear solvers. The Journal of Supercomputing, 63 (2013)."},{"key":"e_1_3_2_2_66_1","doi-asserted-by":"publisher","DOI":"10.1145\/1089014.1089017"},{"key":"e_1_3_2_2_67_1","doi-asserted-by":"publisher","DOI":"10.1145\/779359.779361"},{"key":"e_1_3_2_2_68_1","doi-asserted-by":"publisher","DOI":"10.1145\/3577197"},{"key":"e_1_3_2_2_69_1","doi-asserted-by":"publisher","DOI":"10.5281\/zenodo.17706095"},{"key":"e_1_3_2_2_70_1","doi-asserted-by":"crossref","unstructured":"Alycia Lisito Mathieu Faverge Gr\u00e9goire Pichon and Pierre Ramet. 2024. Enhancing Sparse Direct Solver Scalability Through Runtime System Automatic Data Partition. In WAMTA.","DOI":"10.1007\/978-3-031-61763-8_10"},{"key":"e_1_3_2_2_71_1","unstructured":"Weifeng Liu Ang Li Jonathan Hogg Iain S. Duff and Brian Vinter. 2016. A Synchronization-Free Algorithm for Parallel Sparse Triangular Solves. In Euro-Par."},{"key":"e_1_3_2_2_72_1","unstructured":"Weifeng Liu and Brian Vinter. 2014. An Efficient GPU General Sparse Matrix-Matrix Multiplication for Irregular Data. In IPDPS."},{"key":"e_1_3_2_2_73_1","doi-asserted-by":"crossref","unstructured":"Yang Liu Nan Ding Piyush Sao Samuel Williams and Xiaoye Sherry Li. 2023. Unified Communication Optimization Strategies for Sparse Triangular Solver on CPU and GPU Clusters. In SC.","DOI":"10.1145\/3581784.3607092"},{"key":"e_1_3_2_2_74_1","unstructured":"Yuechen Lu Lijie Zeng Tengcheng Wang Xu Fu Wenxuan Li Helin Cheng Dechuang Yang Zhou Jin Marc Casas and Weifeng Liu. 2024. AmgT: Algebraic Multigrid Solver on Tensor Cores. In SC."},{"key":"e_1_3_2_2_75_1","doi-asserted-by":"publisher","DOI":"10.1007\/s42514-023-00151-1"},{"key":"e_1_3_2_2_76_1","unstructured":"Zhengyang Lu Yuyao Niu and Weifeng Liu. 2020. Efficient Block Algorithms for Parallel Sparse Triangular Solve. In ICPP."},{"key":"e_1_3_2_2_77_1","volume-title":"Batched sparse and mixed-precision linear algebra interface for efficient use of GPU hardware accelerators in scientific applications. Future Generation Computer Systems, 160","author":"Luszczek Piotr","year":"2024","unstructured":"Piotr Luszczek, Ahmad Abdelfattah, Hartwig Anzt, Atsushi Suzuki, and Stanimire Tomov. 2024. Batched sparse and mixed-precision linear algebra interface for efficient use of GPU hardware accelerators in scientific applications. Future Generation Computer Systems, 160 (2024)."},{"key":"e_1_3_2_2_78_1","volume-title":"Nikolopoulos","author":"McGregor Robert L.","year":"2005","unstructured":"Robert L. McGregor, Christos D. Antonopoulos, and Dimitrios S. Nikolopoulos. 2005. Scheduling algorithms for effective thread pairing on hybrid multiprocessors. In IPDPS."},{"key":"e_1_3_2_2_79_1","doi-asserted-by":"publisher","DOI":"10.1145\/3633331"},{"key":"e_1_3_2_2_80_1","volume-title":"Distributed Shared-Memory Threads: DSM-Threads. In Workshop on Run-Time Systems for Parallel Programming.","author":"Mueller Frank","year":"2000","unstructured":"Frank Mueller. 2000. Distributed Shared-Memory Threads: DSM-Threads. In Workshop on Run-Time Systems for Parallel Programming."},{"key":"e_1_3_2_2_81_1","doi-asserted-by":"publisher","DOI":"10.1145\/2450136.2450138"},{"key":"e_1_3_2_2_82_1","doi-asserted-by":"crossref","unstructured":"Dimitrios S. Nikolopoulos Theodore S. Papatheodorou Constantine D. Polychronopoulos Jes\u00fas Labarta and Eduard Ayguad\u00e9. 2000. Is Data Distribution Necessary in OpenMP? In SC.","DOI":"10.1109\/SC.2000.10025"},{"key":"e_1_3_2_2_83_1","doi-asserted-by":"crossref","unstructured":"Yuyao Niu Zhengyang Lu Haonan Ji Shuhui Song Zhou Jin and Weifeng Liu. 2022. TileSpGEMM: a tiled algorithm for parallel sparse general matrix-matrix multiplication on GPUs. In PPoPP.","DOI":"10.1145\/3503221.3508431"},{"key":"e_1_3_2_2_84_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.jpdc.2009.09.007"},{"key":"e_1_3_2_2_85_1","doi-asserted-by":"crossref","unstructured":"Keshav Pingali Donald Nguyen Milind Kulkarni Martin Burtscher M. Amber Hassaan Rashid Kaleem Tsung-Hsien Lee Andrew Lenharth Roman Manevich Mario M\u00e9ndez-Lojo Dimitrios Prountzos and Xin Sui. 2011. The tao of parallelism in algorithms. In PLDI.","DOI":"10.1145\/1993498.1993501"},{"key":"e_1_3_2_2_86_1","unstructured":"Vignesh T. Ravi Michela Becchi Wei Jiang Gagan Agrawal and Srimat Chakradhar. 2012. Scheduling Concurrent Applications on a Cluster of CPU-GPU Nodes. In CCGRID."},{"key":"e_1_3_2_2_87_1","volume-title":"MTM: Rethinking Memory Profiling and Migration for Multi-Tiered Large Memory. In EuroSys.","author":"Ren Jie","year":"2024","unstructured":"Jie Ren, Dong Xu, Junhee Ryu, Kwangsik Shin, Daewoo Kim, and Dong Li. 2024. MTM: Rethinking Memory Profiling and Migration for Multi-Tiered Large Memory. In EuroSys."},{"key":"e_1_3_2_2_88_1","volume-title":"Caracal: A GPU-Resident Sparse LU Solver with Lightweight Fine-Grained Scheduling. In SC.","author":"Ren Jie","year":"2025","unstructured":"Jie Ren, Tingxuan Zhong, Yuxi Hong, Guofeng Feng, Xincheng Wang, Weile Jia, Hatem Ltaief, and David Elliot Keyes. 2025. Caracal: A GPU-Resident Sparse LU Solver with Lightweight Fine-Grained Scheduling. In SC."},{"key":"e_1_3_2_2_89_1","article-title":"Efficient sparse matrix factorization for circuit simulation on vector supercomputers","volume":"8","author":"Visvanathan Sadayappan","year":"1989","unstructured":"Ponnuswamy. Sadayappan and V. Visvanathan. 1989. Efficient sparse matrix factorization for circuit simulation on vector supercomputers. IEEE Transactions on Computer-Aided Design of Integrated Circuits and Systems, 8, 12 (1989).","journal-title":"IEEE Transactions on Computer-Aided Design of Integrated Circuits and Systems"},{"key":"e_1_3_2_2_90_1","volume-title":"Xiaoye Sherry Li, and Richard Vuduc","author":"Sao Piyush","year":"2019","unstructured":"Piyush Sao, Ramakrishnan Kannan, Xiaoye Sherry Li, and Richard Vuduc. 2019. A communication-avoiding 3D sparse triangular solver. In ICS."},{"key":"e_1_3_2_2_91_1","volume-title":"A communication-avoiding 3D algorithm for sparse LU factorization on heterogeneous systems. J. Parallel and Distrib. Comput., 131","author":"Sao Piyush","year":"2019","unstructured":"Piyush Sao, Xiaoye S. Li, and Richard Vuduc. 2019. A communication-avoiding 3D algorithm for sparse LU factorization on heterogeneous systems. J. Parallel and Distrib. Comput., 131 (2019)."},{"key":"e_1_3_2_2_92_1","unstructured":"Piyush Sao Xing Liu Richard Vuduc and Xiaoye Li. 2015. A Sparse Direct Solver for Distributed Memory Xeon Phi-Accelerated Systems. In IPDPS."},{"key":"e_1_3_2_2_93_1","volume-title":"Solving unsymmetric sparse systems of linear equations with PARDISO. Future Generation Computer Systems, 20, 3","author":"Schenk Olaf","year":"2004","unstructured":"Olaf Schenk and Klaus G\u00e4rtner. 2004. Solving unsymmetric sparse systems of linear equations with PARDISO. Future Generation Computer Systems, 20, 3 (2004)."},{"key":"e_1_3_2_2_94_1","doi-asserted-by":"publisher","DOI":"10.1145\/3731599.3767440"},{"key":"e_1_3_2_2_95_1","doi-asserted-by":"crossref","unstructured":"Shumpei Shiina and Kenjiro Taura. 2022. Distributed Continuation Stealing is More Scalable than You Might Think. In CLUSTER.","DOI":"10.1109\/CLUSTER51413.2022.00027"},{"key":"e_1_3_2_2_96_1","volume-title":"Task Bench: A Parameterized Benchmark for Evaluating Parallel Runtime Performance. In SC.","author":"Slaughter Elliott","year":"2020","unstructured":"Elliott Slaughter, Wei Wu, Yuankun Fu, Legend Brandenburg, Nicolai Garcia, Wilhem Kautz, Emily Marx, Kaleb S. Morris, Qinglei Cao, George Bosilca, Seema Mirchandaney, Wonchan Leek, Sean Treichlerk, Patrick McCormick, and Alex Aiken. 2020. Task Bench: A Parameterized Benchmark for Evaluating Parallel Runtime Performance. In SC."},{"key":"e_1_3_2_2_97_1","unstructured":"Rupanshu Soi Michael Bauer Sean Treichler Manolis Papadakis Wonchan Lee Patrick McCormick Alex Aiken and Elliott Slaughter. 2021. Index launches: scalable flexible representation of parallel task groups. In SC."},{"key":"e_1_3_2_2_98_1","doi-asserted-by":"publisher","DOI":"10.1177\/1094342009347710"},{"key":"e_1_3_2_2_99_1","doi-asserted-by":"crossref","unstructured":"James D. Trotter Sinan Ekmek\u00e7iba\u015f\u0131 Johannes Langguth Tugba Torun Emre D\u00fczak\u2208 Aleksandar Ilic and Didem Unat. 2023. Bringing Order to Sparsity: A Sparse Matrix Reordering Study on Multicore CPUs. In SC.","DOI":"10.1145\/3581784.3607046"},{"key":"e_1_3_2_2_100_1","volume-title":"Vetter and Frank Mueller","author":"Jeffrey","year":"2002","unstructured":"Jeffrey S. Vetter and Frank Mueller. 2002. Communication Characteristics of Large-Scale Scientific Applications for Contemporary Cluster Architectures. In IPDPS."},{"key":"e_1_3_2_2_101_1","unstructured":"Tengcheng Wang Wenhao Li Haojie Pei Yuying Sun Zhou Jin and Weifeng Liu. 2023. Accelerating Sparse LU Factorization with Density-Aware Adaptive Matrix Multiplication for Circuit Simulation. In DAC."},{"key":"e_1_3_2_2_102_1","doi-asserted-by":"crossref","unstructured":"Yusheng Weijiang Shruthi Balakrishna Jianqiao Liu and Milind Kulkarni. 2015. Tree dependence analysis. In PLDI.","DOI":"10.1145\/2737924.2737972"},{"key":"e_1_3_2_2_103_1","volume-title":"Bennett","author":"Whitlock Matthew","year":"2018","unstructured":"Matthew Whitlock, Hemanth Kolla, Sean Treichler, Philippe P\u00e9bay, and Janine C. Bennett. 2018. Scalable Collectives for Distributed Asynchronous Many-Task Runtimes. In IPDPSW."},{"key":"e_1_3_2_2_104_1","doi-asserted-by":"crossref","unstructured":"Kai Wu Jie Ren and Dong Li. 2018. Runtime Data Management on Non-Volatile Memory-based Heterogeneous Memory for Task-Parallel Programs. In SC.","DOI":"10.1109\/SC.2018.00034"},{"key":"e_1_3_2_2_105_1","doi-asserted-by":"crossref","unstructured":"Yang Xia Peng Jiang Gagan Agrawal and Rajiv Ramnath. 2023. End-to-End LU Factorization of Large Matrices on GPUs. In PPoPP.","DOI":"10.1145\/3572848.3577486"},{"key":"e_1_3_2_2_106_1","volume-title":"Li","author":"Yamazaki Ichitaro","year":"2012","unstructured":"Ichitaro Yamazaki and Xiaoye S. Li. 2012. New Scheduling Strategies and Hybrid Programming for a Parallel Right-looking Sparse LU Factorization Algorithm on Multicore Cluster Systems. In IPDPS."},{"key":"e_1_3_2_2_107_1","doi-asserted-by":"crossref","unstructured":"Ichitaro Yamazaki Sivasankaran Rajamanickam and Nathan Ellingwood. 2020. Performance Portable Supernode-based Sparse Triangular Solver for Manycore Architectures. In ICPP.","DOI":"10.1145\/3404397.3404428"},{"key":"e_1_3_2_2_108_1","volume-title":"Mille-feuille: A Tile-Grained Mixed Precision Single-Kernel Conjugate Gradient Solver on GPUs. In SC.","author":"Yang Dechuang","year":"2024","unstructured":"Dechuang Yang, Yuxuan Zhao, Yiduo Niu, Weile Jia, En Shao, Weifeng Liu, Guangming Tan, and Zhou Jin. 2024. Mille-feuille: A Tile-Grained Mixed Precision Single-Kernel Conjugate Gradient Solver on GPUs. In SC."},{"key":"e_1_3_2_2_109_1","article-title":"Algorithm 980: Sparse QR Factorization on the GPU","volume":"44","author":"Yeralan Sencer Nuri","year":"2017","unstructured":"Sencer Nuri Yeralan, Timothy A. Davis, Wissam M. Sid-Lakhdar, and Sanjay Ranka. 2017. Algorithm 980: Sparse QR Factorization on the GPU. ACM Trans. Math. Software, 44, 2 (2017).","journal-title":"ACM Trans. Math. Software"},{"key":"e_1_3_2_2_110_1","volume-title":"SFLU: Synchronization-Free Sparse LU Factorization for Fast Circuit Simulation on GPUs. In DAC.","author":"Zhao Jianqi","year":"2022","unstructured":"Jianqi Zhao, Yao Wen, Yuchen Luo, Zhou Jin, Weifeng Liu, and Zhenya Zhou. 2022. SFLU: Synchronization-Free Sparse LU Factorization for Fast Circuit Simulation on GPUs. In DAC."},{"key":"e_1_3_2_2_111_1","volume-title":"Linear solvers for power grid optimization problems: A review of GPU-accelerated linear solvers. Parallel Comput., 111","author":"\u015awirydowicz Kasia","year":"2022","unstructured":"Kasia \u015awirydowicz, Eric Darve, Wesley Jones, Jonathan Maack, Shaked Regev, Michael A. Saunders, Stephen J. Thomas, and Slaven Pele\u0161. 2022. Linear solvers for power grid optimization problems: A review of GPU-accelerated linear solvers. Parallel Comput., 111 (2022)."}],"event":{"name":"PPoPP '26: 31st ACM SIGPLAN Annual Symposium on Principles and Practice of Parallel Programming","location":"Sydney NSW Australia","acronym":"PPoPP '26","sponsor":["SIGHPC ACM Special Interest Group on High Performance Computing, Special Interest Group on High Performance Computing","SIGPLAN ACM Special Interest Group on Programming Languages"]},"container-title":["Proceedings of the 31st ACM SIGPLAN Annual Symposium on Principles and Practice of Parallel Programming"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3774934.3786442","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,1,28]],"date-time":"2026-01-28T15:27:36Z","timestamp":1769614056000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3774934.3786442"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,1,28]]},"references-count":111,"alternative-id":["10.1145\/3774934.3786442","10.1145\/3774934"],"URL":"https:\/\/doi.org\/10.1145\/3774934.3786442","relation":{},"subject":[],"published":{"date-parts":[[2026,1,28]]},"assertion":[{"value":"2026-01-28","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}