{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,19]],"date-time":"2026-06-19T16:12:50Z","timestamp":1781885570358,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":40,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,8,12]],"date-time":"2024-08-12T00:00:00Z","timestamp":1723420800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"funder":[{"DOI":"10.13039\/501100006374","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["U2242210"],"award-info":[{"award-number":["U2242210"]}],"id":[{"id":"10.13039\/501100006374","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,8,12]]},"DOI":"10.1145\/3673038.3673040","type":"proceedings-article","created":{"date-parts":[[2024,8,8]],"date-time":"2024-08-08T18:29:01Z","timestamp":1723141741000},"page":"52-62","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":6,"title":["FP16 Acceleration in Structured Multigrid Preconditioner for Real-World Applications"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-7179-6593","authenticated-orcid":false,"given":"Yi","family":"Zong","sequence":"first","affiliation":[{"name":"Tsinghua University, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-3801-2832","authenticated-orcid":false,"given":"Peinan","family":"Yu","sequence":"additional","affiliation":[{"name":"Tsinghua University, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-1786-9821","authenticated-orcid":false,"given":"Haopeng","family":"Huang","sequence":"additional","affiliation":[{"name":"Tsinghua University, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9740-6581","authenticated-orcid":false,"given":"Wei","family":"Xue","sequence":"additional","affiliation":[{"name":"Tsinghua University, China and Qinghai University, Qinghai Provincial Laboratory for Intelligent Computing and Application, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2024,8,12]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"Ahmad Abdelfattah and et al. 2021. A survey of numerical linear algebra methods utilizing mixed-precision arithmetic. 35 (Mar 2021)."},{"key":"e_1_3_2_1_2_1","volume-title":"GRAPES Numerical Weather Prediction System. Retrieved","author":"China\u00a0Meteorological Administration","year":"2023","unstructured":"China\u00a0Meteorological Administration. 2016. GRAPES Numerical Weather Prediction System. Retrieved July 7, 2023 from https:\/\/www.cma.gov.cn\/2011xwzx\/2011xqxxw\/2011xqxyw\/202110\/t20211030_4079298.html"},{"key":"e_1_3_2_1_3_1","volume-title":"HPL-MXP mixed-precision benchmark. Retrieved","author":"Innovative Computing\u00a0Laboratory at\u00a0University","year":"2023","unstructured":"Innovative Computing\u00a0Laboratory at\u00a0University\u00a0of Tennessee. 2023. HPL-MXP mixed-precision benchmark. Retrieved March 3, 2023 from https:\/\/hpl-mxp.org\/"},{"key":"e_1_3_2_1_4_1","volume-title":"Davis and Yifan Hu","author":"A.","year":"2011","unstructured":"Timothy\u00a0A. Davis and Yifan Hu. 2011. The University of Florida Sparse Matrix Collection. 38, 1, Article 1 (Dec 2011), 25\u00a0pages."},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.procs.2010.04.020"},{"key":"e_1_3_2_1_6_1","first-page":"3","article-title":"Falgout and Jacob\u00a0B. Schroder. 2014","volume":"36","year":"2014","unstructured":"Robert\u00a0D. Falgout and Jacob\u00a0B. Schroder. 2014. Non-Galerkin Coarse Grids for Algebraic Multigrid. 36, 3 (Jan 2014), C309\u2013C334.","journal-title":"Non-Galerkin Coarse Grids for Algebraic Multigrid."},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"crossref","unstructured":"Hormozd Gahvari and et al. 2012. Modeling the Performance of an Algebraic Multigrid Cycle Using Hybrid MPI\/OpenMP. In. 128\u2013137.","DOI":"10.1109\/ICPP.2012.41"},{"key":"e_1_3_2_1_8_1","volume-title":"Glimberg and et al","author":"L.","year":"2013","unstructured":"S.\u00a0L. Glimberg and et al. 2013. A Fast GPU-Accelerated Mixed-Precision Strategy for Fully Nonlinear Water Wave Computations. In. Berlin, Heidelberg, 645\u2013652."},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPDS.2010.61"},{"key":"e_1_3_2_1_10_1","volume-title":"Higham and Theo Mary","author":"J.","year":"2022","unstructured":"Nicholas\u00a0J. Higham and Theo Mary. 2022. Mixed precision algorithms in numerical linear algebra. 31 (2022), 347\u2013414."},{"key":"e_1_3_2_1_11_1","unstructured":"Nhut-Minh Ho and et al. 2017. Exploiting half precision arithmetic in Nvidia GPUs. In."},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"crossref","unstructured":"X. Huang and et al. 2016. P-CSI v1.0 an accelerated barotropic solver for the high-resolution ocean model component in the Community Earth System Model v2.0. 9 11 (2016) 4209\u20134225.","DOI":"10.5194\/gmd-9-4209-2016"},{"key":"e_1_3_2_1_13_1","volume-title":"BFLOAT16 - hardware numerics definition.Retrieved","year":"2023","unstructured":"Intel. 2018. BFLOAT16 - hardware numerics definition.Retrieved Nov 30, 2023 from https:\/\/www.intel.com\/content\/dam\/develop\/external\/us\/en\/documents\/bf16-hardware-numerics-definition-white-paper.pdf"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.advengsoft.2008.11.010"},{"key":"e_1_3_2_1_15_1","volume-title":"Documentation for hypre. Retrieved","author":"Lawrence Livermore\u00a0National Lab. 2023.","year":"2023","unstructured":"Lawrence Livermore\u00a0National Lab. 2023. Documentation for hypre. Retrieved March 3, 2023 from https:\/\/hypre.readthedocs.io\/en\/latest"},{"key":"e_1_3_2_1_16_1","volume-title":"Structured multigrid in HYPRE. Retrieved","author":"Lawrence Livermore\u00a0National Lab. 2023.","year":"2023","unstructured":"Lawrence Livermore\u00a0National Lab. 2023. Structured multigrid in HYPRE. Retrieved March 3, 2023 from https:\/\/hypre.readthedocs.io\/en\/latest\/solvers-smg-pfmg.html"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.2172\/1764323"},{"key":"e_1_3_2_1_18_1","first-page":"5","article-title":"Lowell and et al. 2013","volume":"35","author":"Daniel","year":"2013","unstructured":"Daniel Lowell and et al. 2013. Stencil-Aware GPU Optimization of Iterative Solvers. 35, 5 (2013), S209\u2013S228.","journal-title":"Stencil-Aware GPU Optimization of Iterative Solvers."},{"key":"e_1_3_2_1_19_1","volume-title":"Algebraic error analysis for mixed-precision multigrid solvers. 43","author":"McCormick F.","year":"2020","unstructured":"Stephen\u00a0F. McCormick, Joseph Benzaken, and Rasmus Tamstorf. 2020. Algebraic error analysis for mixed-precision multigrid solvers. 43 (2020), S392\u2013S419."},{"key":"e_1_3_2_1_20_1","unstructured":"Paulius Micikevicius Sharan Narang Jonah Alben Gregory Diamos Erich Elsen David Garcia Boris Ginsburg Michael Houston Oleksii Kuchaiev Ganesh Venkatesh and Hao Wu. 2018. Mixed Precision Training. arxiv:1710.03740\u00a0[cs.AI]"},{"key":"e_1_3_2_1_21_1","volume-title":"M\u00fcller and et al","author":"H.","year":"2014","unstructured":"Eike\u00a0H. M\u00fcller and et al. 2014. Massively parallel solvers for elliptic partial differential equations in numerical weather and climate prediction. 140, 685 (2014), 2608\u20132624."},{"key":"e_1_3_2_1_22_1","first-page":"5","article-title":"2015","volume":"37","author":"Naumov M.","year":"2015","unstructured":"M. Naumov and et al. 2015. AmgX: A Library for GPU Accelerated Algebraic Multigrid and Preconditioned Iterative Methods. 37, 5 (2015), S602\u2013S626.","journal-title":"AmgX: A Library for GPU Accelerated Algebraic Multigrid and Preconditioned Iterative Methods."},{"key":"e_1_3_2_1_23_1","volume-title":"SPE Comparative Solution Project. Retrieved","author":"Society of Petroleum\u00a0Engineers. 2023.","year":"2023","unstructured":"Society of Petroleum\u00a0Engineers. 2023. SPE Comparative Solution Project. Retrieved March 3, 2023 from https:\/\/www.spe.org\/web\/csp\/datasets\/set02.htm"},{"key":"e_1_3_2_1_24_1","unstructured":"Kyaw\u00a0Linn Oo and Andreas Vogel. 2020. Accelerating Geometric Multigrid Preconditioning with Half-Precision Arithmetic on GPUs. arxiv:2007.07539\u00a0[cs.MS]"},{"key":"e_1_3_2_1_25_1","volume-title":"An Introduction of GRAPES.Retrieved","author":"Meteorological\u00a0News Press China","year":"2023","unstructured":"China Meteorological\u00a0News Press. 2014. An Introduction of GRAPES.Retrieved July 7, 2023 from https:\/\/www.cma.gov.cn\/en\/NewsReleases\/MetInstruments\/201403\/t20140327_241784.html"},{"key":"e_1_3_2_1_26_1","volume-title":"Retrieved","author":"Trilinos","year":"2023","unstructured":"Trilinos project. 2023. MueLu. Retrieved Nov 26, 2023 from https:\/\/trilinos.github.io\/muelu.html"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"crossref","unstructured":"Christian Richter and et al. 2014. GPU-accelerated mixed precision algebraic multigrid preconditioners for discrete elliptic field problems. In. 1\u20132.","DOI":"10.1109\/TMAG.2013.2283099"},{"key":"e_1_3_2_1_28_1","unstructured":"Yousef Saad. 2003. (second ed.). Society for Industrial and Applied Mathematics."},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","unstructured":"Martin\u00a0H. Sadd. 2005.. Academic Press. https:\/\/doi.org\/10.1016\/B978-0-12-605811-6.X5000-3","DOI":"10.1016\/B978-0-12-605811-6.X5000-3"},{"key":"e_1_3_2_1_30_1","unstructured":"K. Stuben. 2000. Algebraic Multigrid (AMG) : An Introduction With Applications."},{"key":"e_1_3_2_1_31_1","volume-title":"Retrieved","author":"MFEM","year":"2023","unstructured":"MFEM team. 2023. MFEM examples. Retrieved Nov 26, 2023 from https:\/\/mfem.org\/examples"},{"key":"e_1_3_2_1_32_1","unstructured":"Ulrich Trottenberg and et al. 2001.. Academic Press San Diego California USA."},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"crossref","unstructured":"Yu-Hsiang\u00a0Mike Tsai and et al. 2023. Three-precision algebraic multigrid on GPUs. 149 (12 2023).","DOI":"10.1016\/j.future.2023.07.024"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","unstructured":"Xiaowen Xu and et al. 2017. Algebraic interface-based coarsening AMG preconditioner for multi-scale sparse matrices with applications to radiation hydrodynamics computation. 24 2 (2017) e2078. https:\/\/doi.org\/10.1002\/nla.2078","DOI":"10.1002\/nla.2078"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","unstructured":"Takateru Yamagishi and et al. 2016. GPU Acceleration of a Non-Hydrostatic Ocean Model with a Multigrid Poisson\/Helmholtz Solver. 80 C (jun 2016) 1658\u20131669. https:\/\/doi.org\/10.1016\/j.procs.2016.05.502","DOI":"10.1016\/j.procs.2016.05.502"},{"key":"e_1_3_2_1_36_1","volume-title":"On long-range interpolation operators for aggressive coarsening. 17, 2-3","author":"Yang Ulrike\u00a0Meier","year":"2010","unstructured":"Ulrike\u00a0Meier Yang. 2010. On long-range interpolation operators for aggressive coarsening. 17, 2-3 (2010), 453\u2013472."},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","unstructured":"Xiaojian Yang and et al. 2023. Optimizing Multi-Grid Computation and Parallelization on Multi-Cores. In. 227\u2013239. https:\/\/doi.org\/10.1145\/3577193.3593726","DOI":"10.1145\/3577193.3593726"},{"key":"e_1_3_2_1_38_1","unstructured":"Chensong Zhang and et al. 2023. OpenCAEPoro. Retrieved March 3 2023 from https:\/\/github.com\/OpenCAEPlus\/OpenCAEPoro\/tree\/main\/examples\/spe10"},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","unstructured":"Qianchao Zhu and et al. 2021. Enabling and Scaling the HPCG Benchmark on the Newest Generation Sunway Supercomputer with 42 Million Heterogeneous Cores. In. ACM Article 57 13\u00a0pages. https:\/\/doi.org\/10.1145\/3458817.3476158","DOI":"10.1145\/3458817.3476158"},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1145\/3627535.3638482"}],"event":{"name":"ICPP '24: the 53rd International Conference on Parallel Processing","location":"Gotland Sweden","acronym":"ICPP '24"},"container-title":["Proceedings of the 53rd International Conference on Parallel Processing"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3673038.3673040","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3673038.3673040","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,9,23]],"date-time":"2025-09-23T17:30:22Z","timestamp":1758648622000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3673038.3673040"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,8,12]]},"references-count":40,"alternative-id":["10.1145\/3673038.3673040","10.1145\/3673038"],"URL":"https:\/\/doi.org\/10.1145\/3673038.3673040","relation":{},"subject":[],"published":{"date-parts":[[2024,8,12]]},"assertion":[{"value":"2024-08-12","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}