{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,16]],"date-time":"2026-06-16T14:53:37Z","timestamp":1781621617339,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":35,"publisher":"ACM","license":[{"start":{"date-parts":[[2020,6,23]],"date-time":"2020-06-23T00:00:00Z","timestamp":1592870400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"University of Houston Faculty Startup","award":["NRUF FS 18 NSM WU"],"award-info":[{"award-number":["NRUF FS 18 NSM WU"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2020,6,23]]},"DOI":"10.1145\/3369583.3392685","type":"proceedings-article","created":{"date-parts":[[2020,6,22]],"date-time":"2020-06-22T03:27:27Z","timestamp":1592796447000},"page":"17-28","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":16,"title":["High Accuracy Matrix Computations on Neural Engines: A Study of QR Factorization and its Applications"],"prefix":"10.1145","author":[{"given":"Shaoshuai","family":"Zhang","sequence":"first","affiliation":[{"name":"University of Houston, Houston, TX, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Elaheh","family":"Baharlouei","sequence":"additional","affiliation":[{"name":"University of Houston, Houston, TX, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Panruo","family":"Wu","sequence":"additional","affiliation":[{"name":"University of Houston, Houston, TX, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2020,6,23]]},"reference":[{"key":"e_1_3_2_2_1_1","doi-asserted-by":"publisher","unstructured":"E. Anderson Z. Bai C. Bischof L. S. Blackford J. Demmel J. Dongarra J. Du Croz A. Greenbaum S. Hammarling A. McKenney and D. Sorensen. 1999. LAPACK Users' Guide. Society for Industrial and Applied Mathematics. https:\/\/doi.org\/10.1137\/1.9780898719604","DOI":"10.1137\/1.9780898719604"},{"key":"e_1_3_2_2_2_1","volume-title":"Communication-Avoiding QR Decomposition for GPUs. In 2011 IEEE International Parallel & Distributed Processing Symposium. IEEE","author":"Anderson Michael","year":"2011","unstructured":"Michael Anderson, Grey Ballard, James Demmel, and Kurt Keutzer. 2011. Communication-Avoiding QR Decomposition for GPUs. In 2011 IEEE International Parallel & Distributed Processing Symposium. IEEE, Anchorage, AK, USA, 48--58. https:\/\/doi.org\/10.1109\/IPDPS.2011.15"},{"key":"e_1_3_2_2_3_1","volume-title":"Barlow and Alicja Smoktunowicz","author":"Jesse","year":"2011","unstructured":"Jesse L. Barlow and Alicja Smoktunowicz. 2011. Reorthogonalized Block Classical Gram--Schmidt. arXiv:1108.4209 [math] (Aug. 2011). http:\/\/arxiv.org\/abs\/1108.4209 arXiv: 1108.4209."},{"key":"e_1_3_2_2_4_1","doi-asserted-by":"publisher","DOI":"10.1137\/0908009"},{"key":"e_1_3_2_2_5_1","doi-asserted-by":"publisher","DOI":"10.1007\/BF01939321"},{"key":"e_1_3_2_2_6_1","doi-asserted-by":"publisher","DOI":"10.1007\/BF01939974"},{"key":"e_1_3_2_2_7_1","doi-asserted-by":"crossref","unstructured":"L. S. Blackford Jaeyoung Choi A. Cleary E. D'Azeuedo J. Demmel I. Dhillon S. Hammarling G. Henry A. Petitet K. Stanley D. Walker R. C. Whaley and Jack Dongarra. 1997. ScaLAPACK user's guide .SIAM.","DOI":"10.1137\/1.9780898719642"},{"key":"e_1_3_2_2_10_1","doi-asserted-by":"publisher","DOI":"10.1137\/17M1122918"},{"key":"e_1_3_2_2_11_1","doi-asserted-by":"publisher","DOI":"10.1145\/3330345.3331057"},{"key":"e_1_3_2_2_13_1","doi-asserted-by":"publisher","DOI":"10.1137\/080731992"},{"key":"e_1_3_2_2_14_1","doi-asserted-by":"publisher","DOI":"10.1007\/978--3--319-06548--9_1"},{"key":"e_1_3_2_2_16_1","unstructured":"Iain S Duff and Serge Gratton. 2006. The Parallel Algorithms Team at CERFACS ."},{"key":"e_1_3_2_2_17_1","doi-asserted-by":"publisher","DOI":"10.1147\/rd.444.0605"},{"key":"e_1_3_2_2_18_1","doi-asserted-by":"publisher","DOI":"10.1007\/s00211-005-0615--4"},{"key":"e_1_3_2_2_19_1","volume-title":"Van Loan","author":"Golub Gene H.","year":"2012","unstructured":"Gene H. Golub and Charles F. Van Loan. 2012. Matrix Computations. JHU Press. https:\/\/books.google.com\/books?id=5U-l8U3P-VUC"},{"key":"e_1_3_2_2_20_1","unstructured":"Ga\u00ebl Guennebaud Benotextasciicircumit Jacob and others. 2010. Eigen v3. (2010). http:\/\/eigen.tuxfamily.org"},{"key":"e_1_3_2_2_21_1","doi-asserted-by":"publisher","DOI":"10.1007\/978--3--319--93698--7_45"},{"key":"e_1_3_2_2_22_1","doi-asserted-by":"crossref","unstructured":"Azzam Haidar Stanimire Tomov Jack Dongarra and Nicholas J Higham. 2018b. Harnessing GPU Tensor Cores for Fast FP16 Arithmetic to Speed up Mixed-Precision Iterative Refinement Solvers. In SC.","DOI":"10.1109\/SC.2018.00050"},{"key":"e_1_3_2_2_23_1","doi-asserted-by":"publisher","DOI":"10.1145\/3148226.3148237"},{"key":"e_1_3_2_2_24_1","volume-title":"Methods of conjugate gradients for solving linear systems","author":"Hestenes Magnus Rudolph","unstructured":"Magnus Rudolph Hestenes and Eduard Stiefel. 1952. Methods of conjugate gradients for solving linear systems. Vol. 49. NBS Washington, DC."},{"key":"e_1_3_2_2_25_1","doi-asserted-by":"publisher","DOI":"10.1137\/1.9780898718027"},{"key":"e_1_3_2_2_26_1","volume-title":"Higham and Theo Mary","author":"Nicholas","year":"2018","unstructured":"Nicholas J. Higham and Theo Mary. 2018. A New Approach to Probabilistic Rounding Error Analysis. Technical Report. The University of Manchester."},{"key":"e_1_3_2_2_27_1","doi-asserted-by":"publisher","DOI":"10.1145\/320941.320947"},{"key":"e_1_3_2_2_28_1","doi-asserted-by":"publisher","DOI":"10.1137\/14M0973773"},{"key":"e_1_3_2_2_29_1","volume-title":"Scarpazza","author":"Jia Zhe","year":"2018","unstructured":"Zhe Jia, Marco Maggioni, Benjamin Staiger, and Daniele P. Scarpazza. 2018. Dissecting the NVIDIA Volta GPU Architecture via Microbenchmarking. (2018). http:\/\/arxiv.org\/abs\/1804.06826 arXiv: 1804.06826."},{"key":"e_1_3_2_2_30_1","volume-title":"ICL-UT-17-06. Innovative Computing Laboratory","author":"Kurzak Jakub","unstructured":"Jakub Kurzak, Panruo Wu, Mark Gates, Ichitaro Yamazaki, Piotr Luszczek, Gerald Ragghianti, and Jack Dongarra. 2017.Designing SLATE: Software for Linear Algebra Targeting Exascale. SLATE Working Notes 3, ICL-UT-17-06. Innovative Computing Laboratory, University of Tennessee."},{"key":"e_1_3_2_2_31_1","volume-title":"IPDPSW 2018(2018)","author":"Markidis Stefano","year":"2018","unstructured":"Stefano Markidis, Steven Wei Der Chien, Erwin Laure, Ivy Bo Peng, and Jeffrey S. Vetter. 2018. NVIDIA tensor core programmability, performance & precision.Proceedings - 2018 IEEE 32nd International Parallel and Distributed Processing Symposium Workshops, IPDPSW 2018(2018), 522?531. https:\/\/doi.org\/10.1109\/IPDPSW.2018.00091 arXiv: 1803.04014 ISBN: 9781538655559."},{"key":"e_1_3_2_2_33_1","volume-title":"TSQR on Tensor Cores. SC '19, 29 The International Conference for High Performance Computing, Networking, Storage, and Analysis","author":"Ootomo Hiroyuki","year":"2019","unstructured":"Hiroyuki Ootomo. and Rio Yokota. 2019. TSQR on Tensor Cores. SC '19, 29 The International Conference for High Performance Computing, Networking, Storage, and Analysis (2019). https:\/\/doi.org\/10.1145\/1122445.1122456"},{"key":"e_1_3_2_2_34_1","doi-asserted-by":"publisher","DOI":"10.1145\/355984.355989"},{"key":"e_1_3_2_2_35_1","doi-asserted-by":"publisher","DOI":"10.1145\/2427023.2427030"},{"key":"e_1_3_2_2_36_1","doi-asserted-by":"publisher","DOI":"10.1137\/0910005"},{"key":"e_1_3_2_2_37_1","doi-asserted-by":"publisher","DOI":"10.1137\/1.9781611971408"},{"key":"e_1_3_2_2_38_1","doi-asserted-by":"publisher","DOI":"10.1137\/070682563"},{"key":"e_1_3_2_2_39_1","doi-asserted-by":"publisher","DOI":"10.1137\/1.9780898719574"},{"key":"e_1_3_2_2_40_1","doi-asserted-by":"publisher","DOI":"10.1016\/0024--3795(94)90493--6"}],"event":{"name":"HPDC '20: The 29th International Symposium on High-Performance Parallel and Distributed Computing","location":"Stockholm Sweden","acronym":"HPDC '20","sponsor":["University of Arizona University of Arizona","SIGHPC ACM Special Interest Group on High Performance Computing, Special Interest Group on High Performance Computing","SIGARCH ACM Special Interest Group on Computer Architecture"]},"container-title":["Proceedings of the 29th International Symposium on High-Performance Parallel and Distributed Computing"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3369583.3392685","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3369583.3392685","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T23:44:58Z","timestamp":1750203898000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3369583.3392685"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2020,6,23]]},"references-count":35,"alternative-id":["10.1145\/3369583.3392685","10.1145\/3369583"],"URL":"https:\/\/doi.org\/10.1145\/3369583.3392685","relation":{},"subject":[],"published":{"date-parts":[[2020,6,23]]},"assertion":[{"value":"2020-06-23","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}