{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,2]],"date-time":"2026-07-02T13:43:39Z","timestamp":1782999819037,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":26,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,7,5]],"date-time":"2026-07-05T00:00:00Z","timestamp":1783209600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,7,6]]},"DOI":"10.1145\/3797905.3807857","type":"proceedings-article","created":{"date-parts":[[2026,7,2]],"date-time":"2026-07-02T11:50:37Z","timestamp":1782993037000},"page":"1245-1258","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Cheetah: Optimizing Execution Pipelines for Matrix-Free Finite Element Operators on GPUs"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-3332-2092","authenticated-orcid":false,"given":"Jie","family":"Ren","sequence":"first","affiliation":[{"name":"Computer, Electrical and Mathematical Sciences and Engineering, King Abdullah University of Science and Technology, Thuwal, Western Province, Saudi Arabia"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6897-1095","authenticated-orcid":false,"given":"Hatem","family":"Ltaief","sequence":"additional","affiliation":[{"name":"Computer, Electrical and Mathematical Sciences and Engineering, King Abdullah University of Science and Technology, Thuwal, Western Province, Saudi Arabia"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0435-0433","authenticated-orcid":false,"given":"Stefano","family":"Zampini","sequence":"additional","affiliation":[{"name":"Computer, Electrical and Mathematical Sciences and Engineering, King Abdullah University of Science and Technology, Thuwal, Western Province, Saudi Arabia"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4052-7224","authenticated-orcid":false,"given":"David E.","family":"Keyes","sequence":"additional","affiliation":[{"name":"Computer, Electrical and Mathematical Sciences and Engineering, King Abdullah University of Science and Technology, Thuwal, Western Province, Saudi Arabia"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,7,5]]},"reference":[{"key":"e_1_3_3_1_2_2","doi-asserted-by":"publisher","unstructured":"Robert Anderson Julian Andrej Andrew Barker Jamie Bramwell Jean-Sylvain Camier Jakub Cerveny Veselin Dobrev Yohann Dudouit Aaron Fisher Tzanio Kolev et\u00a0al. 2021. MFEM: A modular finite element methods library. Computers & Mathematics with Applications 81 (2021) 42\u201374. 10.1016\/j.camwa.2020.06.009","DOI":"10.1016\/j.camwa.2020.06.009"},{"key":"e_1_3_3_1_3_2","doi-asserted-by":"publisher","unstructured":"Daniel Arndt Wolfgang Bangerth Maximilian Bergbauer Marco Feder Marc Fehling Johannes Heinz Timo Heister Luca Heltai Martin Kronbichler Matthias Maier et\u00a0al. 2023. The deal.II library version 9.5. Journal of Numerical Mathematics 31 3 (2023) 231\u2013246. 10.1515\/jnma-2023-0089","DOI":"10.1515\/jnma-2023-0089"},{"key":"e_1_3_3_1_4_2","doi-asserted-by":"publisher","unstructured":"W. Bangerth R. Hartmann and G. Kanschat. 2007. deal.II\u2014A general-purpose object-oriented finite element library. ACM Trans. Math. Softw. 33 4 (Aug. 2007) 24\u2013es. 10.1145\/1268776.1268779","DOI":"10.1145\/1268776.1268779"},{"key":"e_1_3_3_1_5_2","doi-asserted-by":"publisher","unstructured":"Valeria Barra Jed Brown Jeremy Thompson and Yohann Dudouit. 2020. High-performance operator evaluations with ease of use: libCEED\u2019s Python interface. SciPy (2020). 10.25080\/Majora-342d178e-00c","DOI":"10.25080\/Majora-342d178e-00c"},{"key":"e_1_3_3_1_6_2","doi-asserted-by":"publisher","unstructured":"Maximilian Bergbauer Peter Munch Wolfgang\u00a0A Wall and Martin Kronbichler. 2025. High-performance matrix-free unfitted finite element operator evaluation. SIAM Journal on Scientific Computing 47 3 (2025) B665\u2013B689. 10.1137\/24M1653689","DOI":"10.1137\/24M1653689"},{"key":"e_1_3_3_1_7_2","doi-asserted-by":"publisher","unstructured":"Jed Brown. 2010. Efficient nonlinear solvers for nodal high-order finite elements in 3D. Journal of Scientific Computing 45 1 (2010) 48\u201363. 10.1007\/s10915-010-9396-8","DOI":"10.1007\/s10915-010-9396-8"},{"key":"e_1_3_3_1_8_2","doi-asserted-by":"publisher","unstructured":"Jed Brown Ahmad Abdelfattah Valeria Barra Natalie Beams Jean-Sylvain Camier Veselin Dobrev Yohann Dudouit Leila Ghaffari Tzanio Kolev David Medina et\u00a0al. 2021. libCEED: Fast algebra for high-order element-based discretizations. Journal of Open Source Software 6 63 (2021) 2945. 10.21105\/joss.02945","DOI":"10.21105\/joss.02945"},{"key":"e_1_3_3_1_9_2","doi-asserted-by":"publisher","unstructured":"Jed Brown Barry Smith and Aron Ahmadia. 2013. Achieving textbook multigrid efficiency for hydrostatic ice sheet flow. SIAM Journal on Scientific Computing 35 2 (2013) B359\u2013B375. 10.1137\/110834512","DOI":"10.1137\/110834512"},{"key":"e_1_3_3_1_10_2","doi-asserted-by":"publisher","DOI":"10.1017\/CBO9780511546792"},{"key":"e_1_3_3_1_11_2","doi-asserted-by":"publisher","unstructured":"Paul Fischer Misun Min Thilina Rathnayake Som Dutta Tzanio Kolev Veselin Dobrev Jean-Sylvain Camier Martin Kronbichler Tim Warburton Kasia \u015awirydowicz et\u00a0al. 2020. Scalability of high-performance PDE solvers. The International Journal of High Performance Computing Applications 34 5 (2020) 562\u2013586. 10.1177\/1094342020915762","DOI":"10.1177\/1094342020915762"},{"key":"e_1_3_3_1_12_2","doi-asserted-by":"publisher","DOI":"10.1109\/SC.2016.83"},{"key":"e_1_3_3_1_13_2","doi-asserted-by":"publisher","unstructured":"Mikl\u00f3s Homolya Lawrence Mitchell Fabio Luporini and David\u00a0A. Ham. 2018. TSFC: A Structure-Preserving Form Compiler. SIAM Journal on Scientific Computing 40 3 (2018) C401\u2013C428. arXiv:https:\/\/doi.org\/10.1137\/17M113064210.1137\/17M1130642","DOI":"10.1137\/17M1130642"},{"key":"e_1_3_3_1_14_2","doi-asserted-by":"publisher","unstructured":"Benjamin\u00a0S Kirk John\u00a0W Peterson Roy\u00a0H Stogner and Graham\u00a0F Carey. 2006. libMesh: a C++ library for parallel adaptive mesh refinement\/coarsening simulations. Engineering with Computers 22 3 (2006) 237\u2013254. 10.1007\/S00366-006-0049-3","DOI":"10.1007\/S00366-006-0049-3"},{"key":"e_1_3_3_1_15_2","doi-asserted-by":"publisher","unstructured":"Tzanio Kolev Paul Fischer Misun Min Jack Dongarra Jed Brown Veselin Dobrev Tim Warburton Stanimire Tomov Mark\u00a0S Shephard Ahmad Abdelfattah et\u00a0al. 2021. Efficient exascale discretizations: High-order finite element methods. The International Journal of High Performance Computing Applications 35 6 (2021) 527\u2013552. 10.1177\/10943420211020803","DOI":"10.1177\/10943420211020803"},{"key":"e_1_3_3_1_16_2","doi-asserted-by":"publisher","unstructured":"Martin Kronbichler and Katharina Kormann. 2012. A generic interface for parallel cell-based finite element operator application. Computers & Fluids 63 (2012) 135\u2013147. 10.1016\/j.compfluid.2012.04.012","DOI":"10.1016\/j.compfluid.2012.04.012"},{"key":"e_1_3_3_1_17_2","doi-asserted-by":"publisher","unstructured":"Martin Kronbichler and Katharina Kormann. 2019. Fast matrix-free evaluation of discontinuous Galerkin finite element operators. ACM Trans. Math. Software 45 3 (2019) 1\u201340. 10.1145\/3325864","DOI":"10.1145\/3325864"},{"key":"e_1_3_3_1_18_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-58667-0_13"},{"key":"e_1_3_3_1_19_2","doi-asserted-by":"publisher","unstructured":"Martin Kronbichler and Karl Ljungkvist. 2019. Multigrid for matrix-free high-order finite element computations on graphics processors. ACM Transactions on Parallel Computing 6 1 (2019) 1\u201332. 10.1145\/3322813","DOI":"10.1145\/3322813"},{"key":"e_1_3_3_1_20_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-14313-2_38"},{"key":"e_1_3_3_1_21_2","first-page":"1","volume-title":"Proceedings of the High Performance Computing Symposium","author":"Ljungkvist Karl","year":"2017","unstructured":"Karl Ljungkvist. 2017. Matrix-free finite-element computations on graphics processors with adaptively refined unstructured meshes. In Proceedings of the High Performance Computing Symposium. 1\u201312."},{"key":"e_1_3_3_1_22_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-23099-8"},{"key":"e_1_3_3_1_23_2","doi-asserted-by":"publisher","unstructured":"James\u00a0W Lottes and Paul\u00a0F Fischer. 2005. Hybrid multigrid\/Schwarz algorithms for the spectral element method. Journal of Scientific Computing 24 1 (2005) 45\u201378. 10.1007\/s10915-004-4787-3","DOI":"10.1007\/s10915-004-4787-3"},{"key":"e_1_3_3_1_24_2","doi-asserted-by":"publisher","unstructured":"Fabio Luporini David\u00a0A. Ham and Paul H.\u00a0J. Kelly. 2017. An Algorithm for the Optimization of Finite Element Integration Loops. ACM Trans. Math. Softw. 44 1 Article 3 (March 2017) 26\u00a0pages. 10.1145\/3054944","DOI":"10.1145\/3054944"},{"key":"e_1_3_3_1_25_2","doi-asserted-by":"publisher","unstructured":"Steven\u00a0A Orszag. 1979. Spectral methods for problems in complex geometries. (1979) 273\u2013305. 10.1016\/B978-0-12-546050-7.50014-9","DOI":"10.1016\/B978-0-12-546050-7.50014-9"},{"key":"e_1_3_3_1_26_2","doi-asserted-by":"publisher","unstructured":"Vasily Volkov and James\u00a0W Demmel. 2008. Benchmarking GPUs to tune dense linear algebra. Proceedings of the ACM\/IEEE Conference on Supercomputing (2008) 1\u201311. 10.1109\/SC.2008.5214359","DOI":"10.1109\/SC.2008.5214359"},{"key":"e_1_3_3_1_27_2","doi-asserted-by":"publisher","unstructured":"Samuel Williams Andrew Waterman and David Patterson. 2009. Roofline: an insightful visual performance model for multicore architectures. Commun. ACM 52 4 (2009) 65\u201376. 10.1145\/1498765.1498785","DOI":"10.1145\/1498765.1498785"}],"event":{"name":"ICS '26: 2026 International Conference on Supercomputing","location":"Belfast United Kingdom","acronym":"ICS '26","sponsor":["SIGHPC ACM Special Interest Group on High Performance Computing, Special Interest Group on High Performance Computing","SIGARCH ACM Special Interest Group on Computer Architecture"]},"container-title":["Proceedings of the 40th ACM International Conference on Supercomputing"],"original-title":[],"deposited":{"date-parts":[[2026,7,2]],"date-time":"2026-07-02T12:50:13Z","timestamp":1782996613000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3797905.3807857"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,5]]},"references-count":26,"alternative-id":["10.1145\/3797905.3807857","10.1145\/3797905"],"URL":"https:\/\/doi.org\/10.1145\/3797905.3807857","relation":{},"subject":[],"published":{"date-parts":[[2026,7,5]]},"assertion":[{"value":"2026-07-05","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}