{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,14]],"date-time":"2026-08-14T21:45:39Z","timestamp":1786743939699,"version":"3.56.0"},"publisher-location":"Cham","reference-count":35,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031971952","type":"print"},{"value":"9783031971969","type":"electronic"}],"license":[{"start":{"date-parts":[[2025,10,1]],"date-time":"2025-10-01T00:00:00Z","timestamp":1759276800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,10,1]],"date-time":"2025-10-01T00:00:00Z","timestamp":1759276800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026]]},"DOI":"10.1007\/978-3-031-97196-9_15","type":"book-chapter","created":{"date-parts":[[2025,9,30]],"date-time":"2025-09-30T22:22:40Z","timestamp":1759270960000},"page":"174-185","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":3,"title":["Leveraging Hardware-Aware Computation in\u00a0Mixed-Precision Matrix Multiply: A Tile-Centric Approach"],"prefix":"10.1007","author":[{"given":"Qiao","family":"Zhang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Rabab","family":"Alomairy","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Dali","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zhuowei","family":"Gu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Qinglei","family":"Cao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2025,10,1]]},"reference":[{"key":"15_CR1","doi-asserted-by":"crossref","unstructured":"Ichimura, T., et al.: A fast scalable implicit solver for nonlinear time-evolution earthquake city problem on low-ordered unstructured finite elements with artificial intelligence and transprecision computing. In: SC18: International Conference for High Performance Computing, Networking, Storage and Analysis, pp. 627\u2013637. IEEE (2018)","DOI":"10.1109\/SC.2018.00052"},{"key":"15_CR2","doi-asserted-by":"crossref","unstructured":"Abdulah, S., et al.: Boosting earth system model outputs and saving petabytes in their storage using exascale climate emulators. ACM Gordon Bell Prize for Climate Model. Finalist (2024)","DOI":"10.1109\/SC41406.2024.00008"},{"key":"15_CR3","unstructured":"Vaswani, A.: Attention is all you need. In: Advances in Neural Information Processing Systems (2017)"},{"key":"15_CR4","unstructured":"Meuer, H., Strohmaier, E., Dongarra, J., Simon, H.: The Top500 List. June 2024. http:\/\/www.top500.org"},{"key":"15_CR5","doi-asserted-by":"crossref","unstructured":"Abdulah, S., et al.: Accelerating geostatistical modeling and prediction with mixed-precision computations: A high-productivity approach with parsec. IEEE Trans. Parall. Distrib. Syst. 33(4), 964\u2013976 (2021)","DOI":"10.1109\/TPDS.2021.3084071"},{"key":"15_CR6","doi-asserted-by":"crossref","unstructured":"Cao, Q., et al.: Reducing data motion and energy consumption of geospatial modeling applications using automated precision conversion. In: 2023 IEEE International Conference on Cluster Computing (CLUSTER), pp. 330\u2013342. IEEE (2023)","DOI":"10.1109\/CLUSTER52292.2023.00035"},{"key":"15_CR7","doi-asserted-by":"publisher","unstructured":"Haidar, A., et al.: The design of fast and energy-efficient linear solvers: on the potential of half-precision arithmetic and iterative refinement techniques. In: Shi, Y., (eds.) ICCS 2018. LNCS, vol. 10860, pp. 586\u2013600. Springer, Cham (2018). https:\/\/doi.org\/10.1007\/978-3-319-93698-7_45","DOI":"10.1007\/978-3-319-93698-7_45"},{"key":"15_CR8","doi-asserted-by":"crossref","unstructured":"Haidar, A., Bayraktar, H., Tomov, S., Dongarra, J., Higham, N.J.: Mixed-precision iterative refinement using tensor cores on gpus to accelerate solution of linear systems. Proc. Royal Soc. A 476(2243), 20200110 (2020)","DOI":"10.1098\/rspa.2020.0110"},{"key":"15_CR9","doi-asserted-by":"publisher","first-page":"104746","DOI":"10.1016\/j.jpdc.2023.104746","volume":"181","author":"A Netti","year":"2023","unstructured":"Netti, A., et al.: Mixed precision support in hpc applications: what about reliability? J. Parall. Distrib. Comput. 181, 104746 (2023)","journal-title":"J. Parall. Distrib. Comput."},{"key":"15_CR10","doi-asserted-by":"crossref","unstructured":"Nandakumar, S.R., et al.: Mixed-precision deep learning based on computational memory. Front. Neurosci. 14, 406 (2020)","DOI":"10.3389\/fnins.2020.00406"},{"key":"15_CR11","doi-asserted-by":"crossref","unstructured":"Walden, A., Nielsen, E., Diskin, B., Zubair, M., et al.: A mixed precision multicolor point-implicit solver for unstructured grids on gpus. In: 2019 IEEE\/ACM 9th Workshop on Irregular Applications: Architectures and Algorithms (IA3), pp. 23\u201330. IEEE (2019)","DOI":"10.1109\/IA349570.2019.00010"},{"key":"15_CR12","doi-asserted-by":"crossref","unstructured":"Brogi, F., Bna, S., Boga, G., Amati, G., Ongaro, T.E., Cerminara, M.: On floating point precision in computational fluid dynamics using openfoam. Future Gener. Comput. Syst. 152, 1\u201316 (2024)","DOI":"10.1016\/j.future.2023.10.006"},{"key":"15_CR13","doi-asserted-by":"crossref","unstructured":"Doucet, N., Ltaief, H., Gratadour, D., Keyes, D.: Mixed-precision tomographic reconstructor computations on hardware accelerators. In: 2019 IEEE\/ACM 9th Workshop on Irregular Applications: Architectures and Algorithms (IA3), pp. 31\u201338. IEEE (2019)","DOI":"10.1109\/IA349570.2019.00011"},{"key":"15_CR14","first-page":"2024","volume":"1\u201328","author":"S Chen","year":"2024","unstructured":"Chen, S., Zhang, Y., Wang, Y., Liu, Z., Li, X., Xue, W.: Mixed-precision computing in the grist dynamical core for weather and climate modelling. Geosci. Model Dev. Discuss. 1\u201328, 2024 (2024)","journal-title":"Geosci. Model Dev. Discuss."},{"key":"15_CR15","doi-asserted-by":"crossref","unstructured":"Jia, W., et al.: Pushing the limit of molecular dynamics with ab initio accuracy to 100 million atoms with machine learning. In: SC20: International Conference for High Performance Computing, Networking, Storage and Analysis, pp. 1\u201314. IEEE (2020)","DOI":"10.1109\/SC41405.2020.00009"},{"key":"15_CR16","doi-asserted-by":"crossref","unstructured":"Boku, T., Ishikawa, K.I., Kuramashi, Y., Meadows, L.: Mixed precision solver scalable to 16000 mpi processes for lattice quantum chromodynamics simulations on the oakforest-pacs system. In: 2017 Fifth International Symposium on Computing and Networking (CANDAR), pp. 362\u2013368. IEEE (2017)","DOI":"10.1109\/CANDAR.2017.69"},{"key":"15_CR17","doi-asserted-by":"crossref","unstructured":"Haidar, A., Tomov, S., Dongarra, J., Higham, N.J.: Harnessing gpu tensor cores for fast fp16 arithmetic to speed up mixed-precision iterative refinement solvers. In: SC18: International Conference for High Performance Computing, Networking, Storage and Analysis, pp. 603\u2013613. IEEE (2018)","DOI":"10.1109\/SC.2018.00050"},{"key":"15_CR18","doi-asserted-by":"crossref","unstructured":"Cao, Q., et al.: Reshaping geostatistical modeling and prediction for extreme-scale environmental applications. In: SC22: International Conference for High Performance Computing, Networking, Storage and Analysis (ACM Gordon Bell Prize Finalist), pp. 1\u201312. IEEE (2022)","DOI":"10.1109\/SC41404.2022.00007"},{"key":"15_CR19","doi-asserted-by":"crossref","unstructured":"Ltaief, H., et al.: Toward capturing genetic epistasis from multivariate genome-wide association studies using mixed-precision kernel ridge regression. ACM Gordon Bell Prize Finalist (2024)","DOI":"10.1109\/SC41406.2024.00012"},{"key":"15_CR20","unstructured":"OpenMP. OpenMP 4.5 Complete Specifications (2015)"},{"issue":"3","key":"15_CR21","doi-asserted-by":"publisher","first-page":"292","DOI":"10.1007\/s10766-009-0101-1","volume":"37","author":"A Duran","year":"2009","unstructured":"Duran, A., Ferrer, R., Ayguade, E., Badia, R.M., Labarta, J.: A proposal to extend the OpenMP tasking model with dependent tasks. Int. J. Parall. Program. 37(3), 292\u2013305 (2009)","journal-title":"Int. J. Parall. Program."},{"key":"15_CR22","unstructured":"Lordan, F., et al.: Servicess: an interoperable programming framework for the cloud. J. Grid Comput. (2014)"},{"key":"15_CR23","doi-asserted-by":"publisher","first-page":"187","DOI":"10.1002\/cpe.1631","volume":"23","author":"C Augonnet","year":"2011","unstructured":"Augonnet, C., Thibault, S., Namyst, R., Wacrenier, P.: StarPU: a unified platform for task scheduling on heterogeneous multicore architectures. Concurrency Computat. Pract. Exper. 23, 187\u2013198 (2011)","journal-title":"Concurrency Computat. Pract. Exper."},{"issue":"2\u20133","key":"15_CR24","doi-asserted-by":"publisher","first-page":"253","DOI":"10.1007\/s00450-012-0217-1","volume":"28","author":"T Heller","year":"2013","unstructured":"Heller, T., Kaiser, H., Iglberger, K.: Application of the ParalleX execution model to stencil-based problems. Comput. Sci. Res. Dev. 28(2\u20133), 253\u2013261 (2013)","journal-title":"Comput. Sci. Res. Dev."},{"key":"15_CR25","doi-asserted-by":"crossref","unstructured":"Bauer, M., Treichler, S., Slaughter, E., Aiken, A.: Legion: expressing locality and independence with logical regions. In: International Conference for High Performance Computing, Networking, Storage and Analysis, SC, pp. 1\u201311. IEEE (2012)","DOI":"10.1109\/SC.2012.71"},{"issue":"1\u20132","key":"15_CR26","doi-asserted-by":"publisher","first-page":"37","DOI":"10.1016\/j.parco.2011.10.003","volume":"38","author":"G Bosilca","year":"2012","unstructured":"Bosilca, G., Bouteiller, A., Danalis, A., H\u00e9rault, T., Lemarinier, P., Dongarra, J.J.: DAGuE: a generic distributed dag engine for high performance computing. Parallel Comput. 38(1\u20132), 37\u201351 (2012)","journal-title":"Parallel Comput."},{"key":"15_CR27","doi-asserted-by":"crossref","unstructured":"Bouteiller, A., Herault, T., Cao, Q., Schuchart, J., Bosilca, G.: PaRSEC: scalability, flexibility, and hybrid architecture support for task-based applications in ECP. Int. J. High Perform. Comput. Appl. (2024)","DOI":"10.1177\/10943420241290520"},{"issue":"8","key":"15_CR28","doi-asserted-by":"publisher","first-page":"1856","DOI":"10.1109\/TPDS.2021.3131657","volume":"33","author":"Q Cao","year":"2021","unstructured":"Cao, Q., Bosilca, G., Losada, N., Wei, W., Zhong, D., Dongarra, J.: Evaluating data redistribution in parsec. IEEE Trans. Parallel Distrib. Syst. 33(8), 1856\u20131872 (2021)","journal-title":"IEEE Trans. Parallel Distrib. Syst."},{"key":"15_CR29","doi-asserted-by":"crossref","unstructured":"Cao, Q., et al.: Leveraging parsec runtime support to tackle challenging 3d data-sparse matrix problems. In: 2021 IEEE International Parallel and Distributed Processing Symposium (IPDPS), pp. 79\u201389. IEEE (2021)","DOI":"10.1109\/IPDPS49936.2021.00017"},{"key":"15_CR30","doi-asserted-by":"crossref","unstructured":"Cao, Q., et al.: A framework to exploit data sparsity in tile low-rank cholesky factorization. In: 2022 IEEE International Parallel and Distributed Processing Symposium (IPDPS), pp. 414\u2013424. IEEE (2022)","DOI":"10.1109\/IPDPS53621.2022.00047"},{"key":"15_CR31","doi-asserted-by":"crossref","unstructured":"Cao, Q., et al.: Performance analysis of tile low-rank cholesky factorization using parsec instrumentation tools. In: 2019 IEEE\/ACM International Workshop on Programming and Performance Visualization Tools (ProTools), pp. 25\u201332. IEEE (2019)","DOI":"10.1109\/ProTools49597.2019.00009"},{"key":"15_CR32","doi-asserted-by":"crossref","unstructured":"Danalis, A., Bosilca, G., Bouteiller, A., Herault, T., Dongarra, J.: PTG: an abstraction for unhindered parallelism. In: 2014 Fourth International Workshop on Domain-Specific Languages and High-Level Frameworks for High Performance Computing, pp. 21\u201330. IEEE (2014)","DOI":"10.1109\/WOLFHPC.2014.8"},{"key":"15_CR33","doi-asserted-by":"crossref","unstructured":"Bosilca, G., et al.: The template task graph (TTG)-an emerging practical dataflow programming paradigm for scientific simulation at extreme scale. In: IEEE\/ACM 5th International Workshop on Extreme Scale Programming Models and Middleware (ESPM2), IEEE (2020)","DOI":"10.1109\/ESPM251964.2020.00011"},{"key":"15_CR34","doi-asserted-by":"crossref","unstructured":"Hoque, R., Herault, T., Bosilca, G., Dongarra, J.: Dynamic task discovery in PaRSEC: a data-flow task-based runtime. In: Proceedings of the 8th Workshop on Latest Advances in Scalable Algorithms for Large-Scale Systems, ScalA 2017 (2017)","DOI":"10.1145\/3148226.3148233"},{"key":"15_CR35","doi-asserted-by":"crossref","unstructured":"Van De Geijn, R.A., Watts, J.: Summa: scalable universal matrix multiplication algorithm. Concurrency: Pract. Exp. 9(4), 255\u2013274 (1997)","DOI":"10.1002\/(SICI)1096-9128(199704)9:4<255::AID-CPE250>3.0.CO;2-2"}],"container-title":["Lecture Notes in Computer Science","Asynchronous Many-Task Systems and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-97196-9_15","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,6,17]],"date-time":"2026-06-17T03:41:31Z","timestamp":1781667691000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-97196-9_15"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,1]]},"ISBN":["9783031971952","9783031971969"],"references-count":35,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-97196-9_15","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,10,1]]},"assertion":[{"value":"1 October 2025","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"WAMTA","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Workshop on Asynchronous Many-Task Systems and Applications","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"St. Louis, MO","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"USA","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2025","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"19 February 2025","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"21 February 2025","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"3","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"wamta2025","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/wamta25.github.io\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}