{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,1,8]],"date-time":"2026-01-08T07:26:34Z","timestamp":1767857194004,"version":"3.49.0"},"publisher-location":"New York, NY, USA","reference-count":28,"publisher":"ACM","license":[{"start":{"date-parts":[[2021,8,9]],"date-time":"2021-08-09T00:00:00Z","timestamp":1628467200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"DOI":"10.13039\/100000001","name":"National Science Foundation","doi-asserted-by":"publisher","award":["1942892"],"award-info":[{"award-number":["1942892"]}],"id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2021,8,9]]},"DOI":"10.1145\/3458744.3474050","type":"proceedings-article","created":{"date-parts":[[2021,9,23]],"date-time":"2021-09-23T16:38:30Z","timestamp":1632415110000},"page":"1-8","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":3,"title":["Accelerating Neural Network Training using Arbitrary Precision Approximating Matrix Multiplication Algorithms"],"prefix":"10.1145","author":[{"given":"Grey","family":"Ballard","sequence":"first","affiliation":[{"name":"Wake Forest University, United States of America"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jack","family":"Weissenberger","sequence":"additional","affiliation":[{"name":"Wake Forest University, United States of America"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Luoping","family":"Zhang","sequence":"additional","affiliation":[{"name":"Wake Forest University, United States of America"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2021,9,23]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1134\/S0081543813070079"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1137\/1.9781611976465.32"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1137\/15M1032168"},{"key":"e_1_3_2_1_4_1","volume-title":"Proceedings of the 20th ACM SIGPLAN Symposium on Principles and Practice of Parallel Programming","author":"R.","year":"2015","unstructured":"Austin\u00a0 R. Benson and Grey Ballard. 2015. A Framework for Practical Parallel Fast Matrix Multiplication . In Proceedings of the 20th ACM SIGPLAN Symposium on Principles and Practice of Parallel Programming ( San Francisco, CA, USA) (PPoPP 2015 ). ACM, New York, NY, USA, 42\u201353. https:\/\/doi.org\/10.1145\/2688500.2688513 10.1145\/2688500.2688513 Austin\u00a0R. Benson and Grey Ballard. 2015. A Framework for Practical Parallel Fast Matrix Multiplication. In Proceedings of the 20th ACM SIGPLAN Symposium on Principles and Practice of Parallel Programming (San Francisco, CA, USA) (PPoPP 2015). ACM, New York, NY, USA, 42\u201353. https:\/\/doi.org\/10.1145\/2688500.2688513"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1007\/BF02575865"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1016\/0020-0190(79)90113-3"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1137\/0209053"},{"key":"e_1_3_2_1_8_1","volume-title":"Advances in Neural Information Processing Systems, H.\u00a0Larochelle, M.\u00a0Ranzato, R.\u00a0Hadsell, M.\u00a0F. Balcan, and H.\u00a0Lin (Eds.), Vol.\u00a033. Curran Associates","author":"Tom Brown","year":"1877","unstructured":"Tom Brown 2020. Language Models are Few-Shot Learners . In Advances in Neural Information Processing Systems, H.\u00a0Larochelle, M.\u00a0Ranzato, R.\u00a0Hadsell, M.\u00a0F. Balcan, and H.\u00a0Lin (Eds.), Vol.\u00a033. Curran Associates , Inc ., 1877 \u20131901. https:\/\/papers.nips.cc\/paper\/2020\/hash\/1457c0d6bfcb4967418bfb8ac142f64a-Abstract.html Tom Brown 2020. Language Models are Few-Shot Learners. In Advances in Neural Information Processing Systems, H.\u00a0Larochelle, M.\u00a0Ranzato, R.\u00a0Hadsell, M.\u00a0F. Balcan, and H.\u00a0Lin (Eds.), Vol.\u00a033. Curran Associates, Inc., 1877\u20131901. https:\/\/papers.nips.cc\/paper\/2020\/hash\/1457c0d6bfcb4967418bfb8ac142f64a-Abstract.html"},{"key":"e_1_3_2_1_9_1","first-page":"0759","article-title":"cuDNN","volume":"1410","author":"Chetlur Sharan","year":"2014","unstructured":"Sharan Chetlur , Cliff Woolley , Philippe Vandermersch , Jonathan Cohen , John Tran , Bryan Catanzaro , and Evan Shelhamer . 2014 . cuDNN : Efficient Primitives for Deep Learning. Technical Report 1410 . 0759 . arXiv. http:\/\/arxiv.org\/abs\/1410.0759 Sharan Chetlur, Cliff Woolley, Philippe Vandermersch, Jonathan Cohen, John Tran, Bryan Catanzaro, and Evan Shelhamer. 2014. cuDNN: Efficient Primitives for Deep Learning. Technical Report 1410.0759. arXiv. http:\/\/arxiv.org\/abs\/1410.0759","journal-title":"Efficient Primitives for Deep Learning. Technical Report"},{"key":"e_1_3_2_1_10_1","volume-title":"Proceedings of the Nineteenth Annual ACM Symposium on Theory of Computing","author":"Coppersmith D.","unstructured":"D. Coppersmith and S. Winograd . 1987. Matrix multiplication via arithmetic progressions . In Proceedings of the Nineteenth Annual ACM Symposium on Theory of Computing ( New York, New York, United States) (STOC \u201987). ACM, New York, NY, USA, 1\u20136. https:\/\/doi.org\/10.1145\/28395.28396 10.1145\/28395.28396 D. Coppersmith and S. Winograd. 1987. Matrix multiplication via arithmetic progressions. In Proceedings of the Nineteenth Annual ACM Symposium on Theory of Computing (New York, New York, United States) (STOC \u201987). ACM, New York, NY, USA, 1\u20136. https:\/\/doi.org\/10.1145\/28395.28396"},{"key":"e_1_3_2_1_12_1","volume-title":"Proceedings of the 32nd International Conference on Machine Learning(ICML \u201915","author":"Gupta Suyog","year":"2015","unstructured":"Suyog Gupta , Ankur Agrawal , Kailash Gopalakrishnan , and Pritish Narayanan . 2015 . Deep Learning with Limited Numerical Precision . In Proceedings of the 32nd International Conference on Machine Learning(ICML \u201915 , Vol.\u00a037), Francis Bachand David Blei (Eds.). PMLR, Lille, France, 1737\u20131746. http:\/\/proceedings.mlr.press\/v37\/gupta15.html Suyog Gupta, Ankur Agrawal, Kailash Gopalakrishnan, and Pritish Narayanan. 2015. Deep Learning with Limited Numerical Precision. In Proceedings of the 32nd International Conference on Machine Learning(ICML \u201915, Vol.\u00a037), Francis Bachand David Blei (Eds.). PMLR, Lille, France, 1737\u20131746. http:\/\/proceedings.mlr.press\/v37\/gupta15.html"},{"key":"e_1_3_2_1_13_1","volume-title":"Multilayer feedforward networks are universal approximators. Neural networks 2, 5","author":"Hornik Kurt","year":"1989","unstructured":"Kurt Hornik , Maxwell Stinchcombe , and Halbert White . 1989. Multilayer feedforward networks are universal approximators. Neural networks 2, 5 ( 1989 ), 359\u2013366. https:\/\/doi.org\/10.1016\/0893-6080(89)90020-8 10.1016\/0893-6080(89)90020-8 Kurt Hornik, Maxwell Stinchcombe, and Halbert White. 1989. Multilayer feedforward networks are universal approximators. Neural networks 2, 5 (1989), 359\u2013366. https:\/\/doi.org\/10.1016\/0893-6080(89)90020-8"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS.2017.56"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1145\/3372419"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.5555\/3122009.3242044"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1145\/3341105.3373852"},{"key":"e_1_3_2_1_19_1","volume-title":"GPUs. In 20th Annual International Conference on High Performance Computing. 139\u2013148","author":"Lai P.","year":"2013","unstructured":"P. Lai , H. Arafat , V. Elango , and P. Sadayappan . 2013. Accelerating Strassen-Winograd\u2019s matrix multiplication algorithm on GPUs. In 20th Annual International Conference on High Performance Computing. 139\u2013148 . https:\/\/doi.org\/10.1109\/HiPC. 2013 .6799109 10.1109\/HiPC.2013.6799109 P. Lai, H. Arafat, V. Elango, and P. Sadayappan. 2013. Accelerating Strassen-Winograd\u2019s matrix multiplication algorithm on GPUs. In 20th Annual International Conference on High Performance Computing. 139\u2013148. https:\/\/doi.org\/10.1109\/HiPC.2013.6799109"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1109\/5.726791"},{"key":"e_1_3_2_1_21_1","volume-title":"Advances in Neural Information Processing Systems, Vol.\u00a028. Curran Associates","author":"Novikov Alexander","year":"2015","unstructured":"Alexander Novikov , Dmitrii Podoprikhin , Anton Osokin , and Dmitry Vetrov . 2015. Tensorizing Neural Networks . In Advances in Neural Information Processing Systems, Vol.\u00a028. Curran Associates , Inc .https:\/\/proceedings.neurips.cc\/paper\/ 2015 \/file\/6855456e2fe46a9d49d3d3af4f57443d-Paper.pdf Alexander Novikov, Dmitrii Podoprikhin, Anton Osokin, and Dmitry Vetrov. 2015. Tensorizing Neural Networks. In Advances in Neural Information Processing Systems, Vol.\u00a028. Curran Associates, Inc.https:\/\/proceedings.neurips.cc\/paper\/2015\/file\/6855456e2fe46a9d49d3d3af4f57443d-Paper.pdf"},{"key":"e_1_3_2_1_22_1","volume-title":"How Can We Speed Up Matrix Multiplication?SIAM Rev. 26, 3","author":"Pan V.","year":"1984","unstructured":"V. Pan . 1984. How Can We Speed Up Matrix Multiplication?SIAM Rev. 26, 3 ( 1984 ), 393\u2013415. https:\/\/doi.org\/10.1137\/1026076 10.1137\/1026076 V. Pan. 1984. How Can We Speed Up Matrix Multiplication?SIAM Rev. 26, 3 (1984), 393\u2013415. https:\/\/doi.org\/10.1137\/1026076"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1137\/0210032"},{"key":"e_1_3_2_1_24_1","volume-title":"Very Deep Convolutional Networks for Large-Scale Image Recognition. In International Conference on Learning Representations. https:\/\/arxiv.org\/abs\/1409","author":"Simonyan Karen","year":"2015","unstructured":"Karen Simonyan and Andrew Zisserman . 2015 . Very Deep Convolutional Networks for Large-Scale Image Recognition. In International Conference on Learning Representations. https:\/\/arxiv.org\/abs\/1409 .1556 Karen Simonyan and Andrew Zisserman. 2015. Very Deep Convolutional Networks for Large-Scale Image Recognition. In International Conference on Learning Representations. https:\/\/arxiv.org\/abs\/1409.1556"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1134\/S0965542513120129"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1134\/S0965542515040168"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1007\/BF02165411"},{"key":"e_1_3_2_1_32_1","volume-title":"Proceedings of Machine Learning and Systems(MLSys \u201920","author":"Wang Yu\u00a0Emma","year":"2020","unstructured":"Yu\u00a0Emma Wang , Gu-Yeon Wei , and David Brooks . 2020 . A Systematic Methodology for Analysis of Deep Learning Hardware and Software Platforms . In Proceedings of Machine Learning and Systems(MLSys \u201920 , Vol.\u00a02), I.\u00a0Dhillon, D.\u00a0Papailiopoulos, and V.\u00a0Sze (Eds.). 30\u201343. https:\/\/proceedings.mlsys.org\/paper\/ 2020\/hash\/c20ad4d76fe97759aa27a0c99bff6710-Abstract.html Yu\u00a0Emma Wang, Gu-Yeon Wei, and David Brooks. 2020. A Systematic Methodology for Analysis of Deep Learning Hardware and Software Platforms. In Proceedings of Machine Learning and Systems(MLSys \u201920, Vol.\u00a02), I.\u00a0Dhillon, D.\u00a0Papailiopoulos, and V.\u00a0Sze (Eds.). 30\u201343. https:\/\/proceedings.mlsys.org\/paper\/2020\/hash\/c20ad4d76fe97759aa27a0c99bff6710-Abstract.html"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1145\/2213977.2214056"},{"key":"e_1_3_2_1_34_1","volume-title":"IEEE International Parallel and Distributed Processing Symposium(IPDPS \u201920)","author":"Yan D.","year":"2020","unstructured":"D. Yan , W. Wang , and X. Chu . 2020. Demystifying Tensor Cores to Optimize Half-Precision Matrix Multiply . In IEEE International Parallel and Distributed Processing Symposium(IPDPS \u201920) . 634\u2013643. https:\/\/doi.org\/10.1109\/IPDPS47924. 2020 .00071 10.1109\/IPDPS47924.2020.00071 D. Yan, W. Wang, and X. Chu. 2020. Demystifying Tensor Cores to Optimize Half-Precision Matrix Multiply. In IEEE International Parallel and Distributed Processing Symposium(IPDPS \u201920). 634\u2013643. https:\/\/doi.org\/10.1109\/IPDPS47924.2020.00071"}],"event":{"name":"ICPP 2021: 50th International Conference on Parallel Processing","location":"Lemont IL USA","acronym":"ICPP 2021"},"container-title":["50th International Conference on Parallel Processing Workshop"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3458744.3474050","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3458744.3474050","content-type":"application\/pdf","content-version":"vor","intended-application":"syndication"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3458744.3474050","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T17:49:06Z","timestamp":1750268946000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3458744.3474050"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021,8,9]]},"references-count":28,"alternative-id":["10.1145\/3458744.3474050","10.1145\/3458744"],"URL":"https:\/\/doi.org\/10.1145\/3458744.3474050","relation":{},"subject":[],"published":{"date-parts":[[2021,8,9]]},"assertion":[{"value":"2021-09-23","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}