{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,16]],"date-time":"2026-07-16T12:00:42Z","timestamp":1784203242033,"version":"3.55.0"},"reference-count":58,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2025,12,20]],"date-time":"2025-12-20T00:00:00Z","timestamp":1766188800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,12,20]],"date-time":"2025-12-20T00:00:00Z","timestamp":1766188800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/100007523","name":"Division of Advanced Cyberinfrastructure","doi-asserted-by":"publisher","award":["2451577"],"award-info":[{"award-number":["2451577"]}],"id":[{"id":"10.13039\/100007523","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["SN COMPUT. SCI."],"DOI":"10.1007\/s42979-025-04575-0","type":"journal-article","created":{"date-parts":[[2025,12,20]],"date-time":"2025-12-20T07:44:40Z","timestamp":1766216680000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":2,"title":["High-Performance Mixed-Precision Matrix Multiplication via Tile-Centric Design on Modern Architectures"],"prefix":"10.1007","volume":"7","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-6690-194X","authenticated-orcid":false,"given":"Qiao","family":"Zhang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Rabab","family":"Alomairy","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Dali","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zhuowei","family":"Gu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Qinglei","family":"Cao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2025,12,20]]},"reference":[{"key":"4575_CR1","unstructured":"Meuer H, Strohmaier E, Dongarra J, Simon, H. The Top500 List; 2024. http:\/\/www.top500.org"},{"issue":"4","key":"4575_CR2","doi-asserted-by":"publisher","first-page":"964","DOI":"10.1109\/TPDS.2021.3084071","volume":"33","author":"S Abdulah","year":"2021","unstructured":"Abdulah S, Cao Q, Pei Y, Bosilca G, Dongarra J, Genton MG, et al. Accelerating geostatistical modeling and prediction with mixed-precision computations: a high-productivity approach with parsec. IEEE Trans Parallel Distrib Syst. 2021;33(4):964\u201376.","journal-title":"IEEE Trans Parallel Distrib Syst"},{"key":"4575_CR3","doi-asserted-by":"crossref","unstructured":"Cao Q, Abdulah S, Ltaief H, Genton M,G, Keyes D, Bosilca G. Reducing data motion and energy consumption of geospatial modeling applications using automated precision conversion. In: 2023 IEEE International Conference on Cluster Computing (CLUSTER), 2023;pp. 330\u2013342. IEEE","DOI":"10.1109\/CLUSTER52292.2023.00035"},{"key":"4575_CR4","doi-asserted-by":"crossref","unstructured":"Haidar A, Abdelfattah A, Zounon M, Wu P, Pranesh S, Tomov S, Dongarra J. The design of fast and energy-efficient linear solvers: on the potential of half-precision arithmetic and iterative refinement techniques. In: Computational Science\u2013ICCS 2018: 18th International Conference, Wuxi, China, 2018;11\u201313, Proceedings, Part I, pp. 586\u2013600 (2018). Springer.","DOI":"10.1007\/978-3-319-93698-7_45"},{"issue":"2243","key":"4575_CR5","doi-asserted-by":"publisher","first-page":"20200110","DOI":"10.1098\/rspa.2020.0110","volume":"476","author":"A Haidar","year":"2020","unstructured":"Haidar A, Bayraktar H, Tomov S, Dongarra J, Higham NJ. Mixed-precision iterative refinement using tensor cores on gpus to accelerate solution of linear systems. Proc R Soc A. 2020;476(2243):20200110.","journal-title":"Proc R Soc A"},{"key":"4575_CR6","doi-asserted-by":"publisher","DOI":"10.1016\/j.jpdc.2023.104746","volume":"181","author":"A Netti","year":"2023","unstructured":"Netti A, Peng Y, Omland P, Paulitsch M, Parra J, Espinosa G, et al. Mixed precision support in hpc applications: what about reliability? J Parallel Distrib Comput. 2023;181:104746.","journal-title":"J Parallel Distrib Comput"},{"key":"4575_CR7","doi-asserted-by":"publisher","first-page":"406","DOI":"10.3389\/fnins.2020.00406","volume":"14","author":"S Nandakumar","year":"2020","unstructured":"Nandakumar S, Le Gallo M, Piveteau C, Joshi V, Mariani G, Boybat I, et al. Mixed-precision deep learning based on computational memory. Front Neurosci. 2020;14:406.","journal-title":"Front Neurosci"},{"key":"4575_CR8","doi-asserted-by":"crossref","unstructured":"Walden A, Nielsen E, Diskin B, Zubair M. A mixed precision multicolor point-implicit solver for unstructured grids on gpus. In: 2019 IEEE\/ACM 9th workshop on irregular applications: architectures and algorithms (IA3), 2019;23\u201330. IEEE.","DOI":"10.1109\/IA349570.2019.00010"},{"key":"4575_CR9","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1016\/j.future.2023.10.006","volume":"152","author":"F Brogi","year":"2024","unstructured":"Brogi F, Bn\u00e0 S, Boga G, Amati G, Ongaro TE, Cerminara M. On floating point precision in computational fluid dynamics using openfoam. Futur Gener Comput Syst. 2024;152:1\u201316.","journal-title":"Futur Gener Comput Syst"},{"key":"4575_CR10","doi-asserted-by":"crossref","unstructured":"Doucet N, Ltaief H, Gratadour D, Keyes D. Mixed-precision tomographic reconstructor computations on hardware accelerators. In: 2019 IEEE\/ACM 9th Workshop on Irregular Applications: Architectures and Algorithms (IA3), 2019;pp. 31\u201338. IEEE.","DOI":"10.1109\/IA349570.2019.00011"},{"key":"4575_CR11","first-page":"1","volume":"2024","author":"S Chen","year":"2024","unstructured":"Chen S, Zhang Y, Wang Y, Liu Z, Li X, Xue W. Mixed-precision computing in the grist dynamical core for weather and climate modelling. Geosci Model Dev Discuss. 2024;2024:1\u201328.","journal-title":"Geosci Model Dev Discuss"},{"key":"4575_CR12","doi-asserted-by":"crossref","unstructured":"Jia W, Wang H, Chen M, Lu D, Lin L, Car R, Weinan E, Zhang L. Pushing the limit of molecular dynamics with ab initio accuracy to 100 million atoms with machine learning. In: SC20: International Conference for High Performance Computing, Networking, Storage and Analysis, 2020;pp. 1\u201314. IEEE.","DOI":"10.1109\/SC41405.2020.00009"},{"key":"4575_CR13","doi-asserted-by":"crossref","unstructured":"Boku T, Ishikawa K-I, Kuramashi Y, Meadows L. Mixed precision solver scalable to 16000 mpi processes for lattice quantum chromodynamics simulations on the oakforest-pacs system. In: 2017 Fifth International Symposium on Computing and Networking (CANDAR), 2017;pp. 362\u2013368. IEEE.","DOI":"10.1109\/CANDAR.2017.69"},{"key":"4575_CR14","doi-asserted-by":"publisher","DOI":"10.1137\/1.9781611977523","volume-title":"Rounding errors in algebraic processes","author":"JH Wilkinson","year":"2023","unstructured":"Wilkinson JH. Rounding errors in algebraic processes. Philadelphia: SIAM; 2023."},{"issue":"2","key":"4575_CR15","doi-asserted-by":"publisher","first-page":"316","DOI":"10.1145\/321386.321394","volume":"14","author":"CB Moler","year":"1967","unstructured":"Moler CB. Iterative refinement in floating point. J ACM (JACM). 1967;14(2):316\u201321.","journal-title":"J ACM (JACM)"},{"issue":"4","key":"4575_CR16","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/1377596.1377597","volume":"34","author":"A Buttari","year":"2008","unstructured":"Buttari A, Dongarra J, Kurzak J, Luszczek P, Tomov S. Using mixed precision for sparse matrix computations to enhance the performance while achieving 64-bit accuracy. ACM Trans Math Softw (TOMS). 2008;34(4):1\u201322.","journal-title":"ACM Trans Math Softw (TOMS)"},{"issue":"4","key":"4575_CR17","doi-asserted-by":"publisher","first-page":"457","DOI":"10.1177\/1094342007084026","volume":"21","author":"A Buttari","year":"2007","unstructured":"Buttari A, Dongarra J, Langou J, Langou J, Luszczek P, Kurzak J. Mixed precision iterative refinement techniques for the solution of dense linear systems. Int J High Perform Comput Appl. 2007;21(4):457\u201366.","journal-title":"Int J High Perform Comput Appl"},{"issue":"12","key":"4575_CR18","doi-asserted-by":"publisher","first-page":"2526","DOI":"10.1016\/j.cpc.2008.11.005","volume":"180","author":"M Baboulin","year":"2009","unstructured":"Baboulin M, Buttari A, Dongarra J, Kurzak J, Langou J, Langou J, et al. Accelerating scientific computations with mixed precision algorithms. Comput Phys Commun. 2009;180(12):2526\u201333.","journal-title":"Comput Phys Commun"},{"key":"4575_CR19","doi-asserted-by":"crossref","unstructured":"Haidar A, Tomov S, Dongarra J, Higham NJ. Harnessing gpu tensor cores for fast fp16 arithmetic to speed up mixed-precision iterative refinement solvers. In: SC18: International Conference for High Performance Computing, Networking, Storage and Analysis, 2018;pp. 603\u2013613. IEEE.","DOI":"10.1109\/SC.2018.00050"},{"key":"4575_CR20","doi-asserted-by":"publisher","DOI":"10.1137\/1.9780898718027","volume-title":"Accuracy and stability of numerical algorithms","author":"NJ Higham","year":"2002","unstructured":"Higham NJ. Accuracy and stability of numerical algorithms. Philadelphia: SIAM; 2002."},{"issue":"4","key":"4575_CR21","doi-asserted-by":"publisher","first-page":"344","DOI":"10.1177\/10943420211003313","volume":"35","author":"A Abdelfattah","year":"2021","unstructured":"Abdelfattah A, Anzt H, Boman EG, Carson E, Cojean T, Dongarra J, et al. A survey of numerical linear algebra methods utilizing mixed-precision arithmetic. Int J High Perform Comput Appl. 2021;35(4):344\u201369.","journal-title":"Int J High Perform Comput Appl"},{"key":"4575_CR22","doi-asserted-by":"crossref","unstructured":"Cao Q, Abdulah S, Alomairy R, Pei Y, Nag P, Bosilca G, Dongarra J, Genton M,G, Keyes D,E, Ltaief H, Sun Y. Reshaping geostatistical modeling and prediction for extreme-scale environmental applications. In: SC22: International Conference for High Performance Computing, Networking, Storage and Analysis (ACM Gordon Bell Prize Finalist), 2022;pp. 1\u201312. IEEE.","DOI":"10.1109\/SC41404.2022.00007"},{"key":"4575_CR23","doi-asserted-by":"crossref","unstructured":"Abdulah S, Baker AH, Bosilca G, Cao Q, Castruccio S, Genton MG, Keyes DE, Khalid Z, Ltaief H, Yan\u00a0Song GLS, Sun Y. Boosting earth system model outputs and saving petabytes in their storage using exascale climate emulators. ACM Gordon Bell Prize for Climate Modelling Finalist, 2024.","DOI":"10.1109\/SC41406.2024.00008"},{"key":"4575_CR24","doi-asserted-by":"crossref","unstructured":"Ltaief H, Alomairy R, Cao Q, Ren J, Slim L, Kurth T, Dorschner B, Bougouffa S, Abdelkhalek R, Keyes DE. Toward capturing genetic epistasis from multivariate genome-wide association studies using mixed-precision kernel ridge regression. ACM Gordon Bell Prize Finalist, 2024.","DOI":"10.1109\/SC41406.2024.00012"},{"key":"4575_CR25","doi-asserted-by":"crossref","unstructured":"Ichimura T, Fujita K, Yamaguchi T, Naruse A, Wells JC, Schulthess TC, Straatsma TP, Zimmer CJ, Martinasso M, Nakajima K, et al. A fast scalable implicit solver for nonlinear time-evolution earthquake city problem on low-ordered unstructured finite elements with artificial intelligence and transprecision computing. In: SC18: International Conference for High Performance Computing, Networking, Storage and Analysis, 2018;pp. 627\u2013637. IEEE.","DOI":"10.1109\/SC.2018.00052"},{"key":"4575_CR26","unstructured":"Vaswani A. Attention is all you need. Advances in Neural Information Processing Systems, 2017."},{"key":"4575_CR27","doi-asserted-by":"publisher","unstructured":"Yu C, Chen T, Gan Z. Boost transformer-based language models with GPU-friendly sparsity and quantization. In: Rogers A, Boyd-Graber J, Okazaki N (eds.) Findings of the Association for Computational Linguistics: ACL 2023, pp. 218\u2013235. Association for Computational Linguistics, Toronto, Canada. 2023. https:\/\/doi.org\/10.18653\/v1\/2023.findings-acl.15.","DOI":"10.18653\/v1\/2023.findings-acl.15"},{"key":"4575_CR28","unstructured":"Srinivasa Kumar PK. Evaluating full int8 quantization and inference techniques for causal language model. Master\u2019s thesis, University of Twente, 2025."},{"key":"4575_CR29","first-page":"196","volume":"6","author":"Y Zhao","year":"2024","unstructured":"Zhao Y, Lin C-Y, Zhu K, Ye Z, Chen L, Zheng S, et al. Atom: Low-bit quantization for efficient and accurate llm serving. Proc Mach Learn Syst. 2024;6:196\u2013209.","journal-title":"Proc Mach Learn Syst"},{"key":"4575_CR30","doi-asserted-by":"publisher","first-page":"187","DOI":"10.1002\/cpe.1631","volume":"23","author":"C Augonnet","year":"2011","unstructured":"Augonnet C, Thibault S, Namyst R, Wacrenier P. StarPU: a unified platform for task scheduling on heterogeneous multicore architectures. Concurr Comput Pract Exp. 2011;23:187\u201398.","journal-title":"Concurr Comput Pract Exp"},{"issue":"3","key":"4575_CR31","doi-asserted-by":"publisher","first-page":"292","DOI":"10.1007\/s10766-009-0101-1","volume":"37","author":"A Duran","year":"2009","unstructured":"Duran A, Ferrer R, Ayguade E, Badia RM, Labarta J. A proposal to extend the OpenMP tasking model with dependent tasks. Int J Parallel Program. 2009;37(3):292\u2013305. https:\/\/doi.org\/10.1007\/s10766-009-0101-1.","journal-title":"Int J Parallel Program"},{"key":"4575_CR32","unstructured":"OpenMP. OpenMP 4.5 Complete Specifications, 2015. https:\/\/www.openmp.org\/wp-content\/uploads\/openmp-4.5.pdf."},{"issue":"2\u20133","key":"4575_CR33","doi-asserted-by":"publisher","first-page":"253","DOI":"10.1007\/s00450-012-0217-1","volume":"28","author":"T Heller","year":"2013","unstructured":"Heller T, Kaiser H, Iglberger K. Application of the ParalleX execution model to stencil-based problems. Comput Sci Res Dev. 2013;28(2\u20133):253\u201361. https:\/\/doi.org\/10.1007\/s00450-012-0217-1.","journal-title":"Comput Sci Res Dev"},{"key":"4575_CR34","doi-asserted-by":"publisher","unstructured":"Bauer M, Treichler S, Slaughter E, Aiken A. Legion: Expressing locality and independence with logical regions. In: International Conference for High Performance Computing, Networking, Storage and Analysis, SC, 2012;pp. 1\u201311. https:\/\/doi.org\/10.1109\/SC.2012.71. IEEE.","DOI":"10.1109\/SC.2012.71"},{"key":"4575_CR35","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1109\/MCSE.2013.98","volume":"99","author":"G Bosilca","year":"2013","unstructured":"Bosilca G, Bouteiller A, Danalis A, Faverge M, Herault T, Dongarra J. PaRSEC: a Programming Paradigm Exploiting Heterogeneity for Enhancing Scalability. Comput Sci Eng. 2013;99:1. https:\/\/doi.org\/10.1109\/MCSE.2013.98.","journal-title":"Comput Sci Eng"},{"key":"4575_CR36","doi-asserted-by":"crossref","unstructured":"Danalis A, Jagode H, Bosilca G, Dongarra J. PaRSEC in practice: optimizing a legacy chemistry application through distributed task-based execution. In: 2015 IEEE International Conference on Cluster Computing, 2015;pp. 304\u2013313.","DOI":"10.1109\/CLUSTER.2015.50"},{"issue":"4","key":"4575_CR37","doi-asserted-by":"publisher","first-page":"540","DOI":"10.1177\/1094342016672543","volume":"32","author":"H Jagode","year":"2017","unstructured":"Jagode H, Danalis A, Dongarra J. Accelerating nwchem coupled cluster through dataflow-based execution. Int J High Perform Comput Appl. 2017;32(4):540\u201351. https:\/\/doi.org\/10.1177\/1094342016672543. (1\u201313).","journal-title":"Int J High Perform Comput Appl"},{"key":"4575_CR38","doi-asserted-by":"crossref","unstructured":"Cao Q, Pei Y, Herault T, Akbudak K, Mikhalev A, Bosilca G, Ltaief H, Keyes D, Dongarra J. Performance analysis of tile low-rank cholesky factorization using parsec instrumentation tools. In: 2019 IEEE\/ACM International Workshop on Programming and Performance Visualization Tools (ProTools), 2019;pp. 25\u201332. IEEE.","DOI":"10.1109\/ProTools49597.2019.00009"},{"key":"4575_CR39","doi-asserted-by":"crossref","unstructured":"Herault T, Robert Y, Bosilca G, Harrison RJ, Lewis CA, Valeev EF, Dongarra JJ. Distributed-memory multi-gpu block-sparse tensor contraction for electronic structure. In: 2021 IEEE International Parallel and Distributed Processing Symposium (IPDPS), 2021;pp. 537\u2013546. IEEE.","DOI":"10.1109\/IPDPS49936.2021.00062"},{"key":"4575_CR40","doi-asserted-by":"crossref","unstructured":"Cao Q, Alomairy R, Pei Y, Bosilca G, Ltaief H, Keyes D, Dongarra J. A framework to exploit data sparsity in tile low-rank cholesky factorization. In: 2022 IEEE International Parallel and Distributed Processing Symposium (IPDPS), 2022;pp. 414\u2013424. IEEE.","DOI":"10.1109\/IPDPS53621.2022.00047"},{"key":"4575_CR41","doi-asserted-by":"crossref","unstructured":"Alomairy R, Cao Q, Ltaief H, Keyes D, Edelman A. Scalable hamming distance computation using accelerated matrix transformations. In: ISC High Performance 2025 Research Paper Proceedings (40th International Conference), 2025;pp. 1\u201313. Prometeus GmbH.","DOI":"10.23919\/ISC.2025.11017727"},{"key":"4575_CR42","doi-asserted-by":"crossref","unstructured":"Zhang Q, Alomairy R, Wang D, Gu Z, Cao Q. Leveraging Hardware-Aware Computation in Mixed-Precision Matrix Multiply: A Tile-Centric Approach, 2025. https:\/\/arxiv.org\/abs\/2508.14848.","DOI":"10.1007\/978-3-031-97196-9_15"},{"key":"4575_CR43","doi-asserted-by":"publisher","unstructured":"Jouppi NP, Young C, Patil N, Patterson DA, Agrawal G, Bajwa R, Bates S, Bhatia S, Boden N, Borchers A, Boyle R, Cantin P-L, Chao C, Clark C, Coriell J, Daley M, Dau M, Dean J, Gelb B, Ghaemmaghami TV, Gottipati R, Gulland W, Hagmann R, Ho CR, Hogberg D, Hu J, Hundt R, Hurt D, Ibarz J, Jaffey A, Jaworski A, Kaplan A, Khaitan H, Koch A, Kumar N, Lacy S, Laudon J, Law J, Le D, Leary C, Liu Z, Lucke K, Lundin A, MacKean G, Maggiore A, Mahony M, Miller K, Nagarajan R, Narayanaswami R, Ni R, Nix K, Norrie T, Omernick M, Penukonda N, Phelps A, Ross J, Salek A, Samadiani E, Severn C, Sizikov G, Snelham M, Souter J, Steinberg D, Swing A, Tan M, Thorson G, Tian B, Toma H, Tuttle E, Vasudevan V, Walter R, Wang W, Wilcox E, Yoon DH. In-datacenter performance analysis of a tensor processing unit. In: Proceedings of the 44th Annual International Symposium on Computer Architecture (ISCA), 2017;pp. 1\u201312. https:\/\/doi.org\/10.1145\/3079856.3080246. IEEE\/ACM.","DOI":"10.1145\/3079856.3080246"},{"key":"4575_CR44","doi-asserted-by":"publisher","unstructured":"Gupta S, Agrawal A, Gopalakrishnan K, Narayanan P. Deep learning with limited numerical precision. CoRR, 2015. https:\/\/doi.org\/10.48550\/arXiv.1502.02551.","DOI":"10.48550\/arXiv.1502.02551"},{"key":"4575_CR45","unstructured":"Micikevicius P, Narang S, Alben J, Diamos G, Elsen E, Garcia D, Ginsburg B, Houston M, Kuchaiev O, Venkatesh G, Wu H. Mixed precision training. In: International Conference on Learning Representations (ICLR), 2018. https:\/\/openreview.net\/forum?id=r1gs9JgRZ."},{"issue":"1","key":"4575_CR46","doi-asserted-by":"publisher","first-page":"67","DOI":"10.1007\/s10723-013-9272-5","volume":"12","author":"F Lordan","year":"2014","unstructured":"Lordan F, Tejedor E, Ejarque J, Rafanell R, Alvarez J, Marozzo F, et al. Servicess: an interoperable programming framework for the cloud. J Grid Comput. 2014;12(1):67\u201391.","journal-title":"J Grid Comput"},{"issue":"1\u20132","key":"4575_CR47","doi-asserted-by":"publisher","first-page":"37","DOI":"10.1016\/j.parco.2011.10.003","volume":"38","author":"G Bosilca","year":"2012","unstructured":"Bosilca G, Bouteiller A, Danalis A, H\u00e9rault T, Lemarinier P, Dongarra JJ. DAGuE: a generic distributed DAG engine for high performance computing. Parallel Comput. 2012;38(1\u20132):37\u201351.","journal-title":"Parallel Comput"},{"issue":"1","key":"4575_CR48","doi-asserted-by":"publisher","first-page":"147","DOI":"10.1177\/10943420241290520","volume":"39","author":"A Bouteiller","year":"2024","unstructured":"Bouteiller A, Herault T, Cao Q, Schuchart J, Bosilca G. PaRSEC: scalability, flexibility, and hybrid architecture support for task-based applications in ECP. Int J High Perform Comput Appl. 2024;39(1):147\u201366.","journal-title":"Int J High Perform Comput Appl"},{"issue":"8","key":"4575_CR49","doi-asserted-by":"publisher","first-page":"1856","DOI":"10.1109\/TPDS.2021.3131657","volume":"33","author":"Q Cao","year":"2021","unstructured":"Cao Q, Bosilca G, Losada N, Wu W, Zhong D, Dongarra J. Evaluating data redistribution in parsec. IEEE Trans Parallel Distrib Syst. 2021;33(8):1856\u201372.","journal-title":"IEEE Trans Parallel Distrib Syst"},{"key":"4575_CR50","doi-asserted-by":"crossref","unstructured":"Cao Q, Pei Y, Akbudak K, Bosilca G, Ltaief H, Keyes D, Dongarra J. Leveraging parsec runtime support to tackle challenging 3d data-sparse matrix problems. In: 2021 IEEE International Parallel and Distributed Processing Symposium (IPDPS), 2021;pp. 79\u201389. IEEE.","DOI":"10.1109\/IPDPS49936.2021.00017"},{"key":"4575_CR51","doi-asserted-by":"crossref","unstructured":"Danalis A, Bosilca G, Bouteiller A, Herault T, Dongarra J. PTG: an abstraction for unhindered parallelism. In: 2014 Fourth International Workshop on Domain-Specific Languages and High-Level Frameworks for High Performance Computing, 2014;pp. 21\u201330. IEEE.","DOI":"10.1109\/WOLFHPC.2014.8"},{"key":"4575_CR52","doi-asserted-by":"crossref","unstructured":"Bosilca G, Harrison RJ, Herault T, Javanmard MM, Nookala P, Valeev EF. The template task graph (TTG)-an emerging practical dataflow programming paradigm for scientific simulation at extreme scale. In: IEEE\/ACM 5th International Workshop on Extreme Scale Programming Models and Middleware (ESPM2), 2020. IEEE.","DOI":"10.1109\/ESPM251964.2020.00011"},{"key":"4575_CR53","doi-asserted-by":"crossref","unstructured":"Hoque R, Herault T, Bosilca G, Dongarra J. Dynamic task discovery in PaRSEC: a data-flow task-based runtime. In: Proceedings of the 8th Workshop on Latest Advances in Scalable Algorithms for Large-Scale Systems. ScalA \u201917, 2017.","DOI":"10.1145\/3148226.3148233"},{"key":"4575_CR54","doi-asserted-by":"crossref","unstructured":"Thomadakis P, Chrisochoides N. Runtime support for performance portability on heterogeneous distributed platforms. CoRR, 2023. arXiv:2303.02543 [cs.DC].","DOI":"10.3389\/fhpcp.2024.1417040"},{"issue":"4","key":"4575_CR55","doi-asserted-by":"publisher","first-page":"255","DOI":"10.1002\/(SICI)1096-9128(199704)9:4<255::AID-CPE250>3.0.CO;2-2","volume":"9","author":"RA Van De Geijn","year":"1997","unstructured":"Van De Geijn RA, Watts J. Summa: scalable universal matrix multiplication algorithm. Concurr Pract Exp. 1997;9(4):255\u201374.","journal-title":"Concurr Pract Exp"},{"key":"4575_CR56","doi-asserted-by":"publisher","unstructured":"Choi J, Demmel J, Dhillon IS, Ostrouchov S, Petitet A, Stanley K, Walker DW, Whaley RC. ScaLAPACK: a portable linear algebra library for distributed memory computers\u2014design issues and performance. In: Dongarra, J, Madsen, K, Wasniewski, JE (eds.) Proceedings of the Second International Workshop on Applied Parallel Computing: Computations in Physics, Chemistry and Engineering Science (PARA \u201995), Lyngby, Denmark, August 21-24, 1995. Lecture Notes in Computer Science, vol. 1041, pp. 95\u2013106. Springer, Berlin, Heidelberg, 1996. https:\/\/doi.org\/10.1007\/3-540-60902-4_12.","DOI":"10.1007\/3-540-60902-4_12"},{"issue":"7","key":"4575_CR57","doi-asserted-by":"publisher","first-page":"2036","DOI":"10.1109\/TPDS.2015.2481890","volume":"27","author":"J Kurzak","year":"2015","unstructured":"Kurzak J, Anzt H, Gates M, Dongarra J. Implementation and tuning of batched cholesky factorization and solve for nvidia gpus. IEEE Trans Parallel Distrib Syst. 2015;27(7):2036\u201348.","journal-title":"IEEE Trans Parallel Distrib Syst"},{"issue":"1","key":"4575_CR58","doi-asserted-by":"publisher","first-page":"38","DOI":"10.1016\/j.parco.2008.10.002","volume":"35","author":"A Buttari","year":"2007","unstructured":"Buttari A, Langou J, Kurzak J, Dongarra J. A class of parallel tiled linear algebra algorithms for multicore architectures. Parallel Comput. 2007;35(1):38\u201353.","journal-title":"Parallel Comput"}],"container-title":["SN Computer Science"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s42979-025-04575-0.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s42979-025-04575-0","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s42979-025-04575-0.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,20]],"date-time":"2025-12-20T07:44:44Z","timestamp":1766216684000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s42979-025-04575-0"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,12,20]]},"references-count":58,"journal-issue":{"issue":"1","published-online":{"date-parts":[[2026,1]]}},"alternative-id":["4575"],"URL":"https:\/\/doi.org\/10.1007\/s42979-025-04575-0","relation":{},"ISSN":["2661-8907"],"issn-type":[{"value":"2661-8907","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,12,20]]},"assertion":[{"value":"18 June 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"25 November 2025","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"20 December 2025","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"Not Applicable.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}},{"value":"Not Applicable.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Research Involving Human and\/or Animals"}},{"value":"Not Applicable.","order":4,"name":"Ethics","group":{"name":"EthicsHeading","label":"Informed Consent"}}],"article-number":"24"}}