{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,18]],"date-time":"2026-08-18T15:21:34Z","timestamp":1787066494332,"version":"3.56.0"},"reference-count":51,"publisher":"Association for Computing Machinery (ACM)","issue":"7","license":[{"start":{"date-parts":[[2020,6,18]],"date-time":"2020-06-18T00:00:00Z","timestamp":1592438400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":["Commun. ACM"],"published-print":{"date-parts":[[2020,6,18]]},"abstract":"<jats:p>DSAs gain efficiency from specialization and performance from parallelism.<\/jats:p>","DOI":"10.1145\/3361682","type":"journal-article","created":{"date-parts":[[2020,6,18]],"date-time":"2020-06-18T16:23:33Z","timestamp":1592497413000},"page":"48-57","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":231,"title":["Domain-specific hardware accelerators"],"prefix":"10.1145","volume":"63","author":[{"given":"William J.","family":"Dally","sequence":"first","affiliation":[{"name":"Stanford University, CA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yatish","family":"Turakhia","sequence":"additional","affiliation":[{"name":"University of California, Santa Cruz, CA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Song","family":"Han","sequence":"additional","affiliation":[{"name":"Massachusetts Institute of Technology, Cambridge, MA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2020,6,18]]},"reference":[{"key":"e_1_2_1_1_1","volume-title":"OSDI","author":"Abadi M.","year":"2016","unstructured":"Abadi , M. , Tensorflow: A system for large-scale machine learning . In OSDI ( 2016 ), 265--283. Abadi, M., et al. Tensorflow: A system for large-scale machine learning. In OSDI (2016), 265--283."},{"key":"e_1_2_1_2_1","first-page":"1","article-title":"A hardware logic simulation system","volume":"9","author":"Agrawal P.","year":"1990","unstructured":"Agrawal , P. , Dally , W.J . A hardware logic simulation system . IEEE TCAD 9 , 1 ( 1990 ), 19--29. Agrawal, P., Dally, W.J. A hardware logic simulation system. IEEE TCAD 9, 1 (1990), 19--29.","journal-title":"IEEE TCAD"},{"key":"e_1_2_1_3_1","volume-title":"Graph, (SIGGRAPH) 96","author":"Beers A.C.","year":"1996","unstructured":"Beers , A.C. , Agrawala , M. , Chaddha , N. Rendering from compressed textures. ACM Trans , Graph, (SIGGRAPH) 96 ( 1996 ), 373--378. Beers, A.C., Agrawala, M., Chaddha, N. Rendering from compressed textures. ACM Trans, Graph, (SIGGRAPH) 96 (1996), 373--378."},{"key":"e_1_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1109\/MM.2018.022071134"},{"key":"e_1_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1109\/MM.2018.022071131"},{"key":"e_1_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1145\/2333660.2333747"},{"key":"e_1_2_1_7_1","first-page":"2","article-title":"Customizable domain-specific computing","volume":"28","author":"Cong J.","year":"2010","unstructured":"Cong , J. , Sarkar , V. , Reinman , G. , Bui , A . Customizable domain-specific computing . IEEE Des. Test Comput. 28 , 2 ( 2010 ), 6--15. Cong, J., Sarkar, V., Reinman, G., Bui, A. Customizable domain-specific computing. IEEE Des. Test Comput. 28, 2 (2010), 6--15.","journal-title":"IEEE Des. Test Comput."},{"key":"e_1_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1145\/173284.155333"},{"key":"e_1_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1186\/s12859-016-0930-z"},{"key":"e_1_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1109\/MC.2008.224"},{"key":"e_1_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCAD.1985.1270120"},{"key":"e_1_2_1_12_1","volume-title":"ISCA","author":"Esmaeilzadeh H.","year":"2011","unstructured":"Esmaeilzadeh , H. , Blem , E. , Amant , R.S. , Dark silicon and the end of multicore scaling . In ISCA ( 2011 ). IEEE , 365--376. Esmaeilzadeh, H., Blem, E., Amant, R.S., et al. Dark silicon and the end of multicore scaling. In ISCA (2011). IEEE, 365--376."},{"key":"e_1_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1007\/BF02700035"},{"key":"e_1_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1016\/0022-2836(82)90398-9"},{"key":"e_1_2_1_15_1","volume-title":"Goya Inference Platform White Paper v1.7","author":"Habana Labs","year":"2019","unstructured":"Habana Labs . Goya Inference Platform White Paper v1.7 , 2019 . https:\/\/tinyurl.com\/yxlcfx54 Habana Labs. Goya Inference Platform White Paper v1.7, 2019. https:\/\/tinyurl.com\/yxlcfx54"},{"key":"e_1_2_1_16_1","volume-title":"Dally","author":"Han S.","year":"2016","unstructured":"Han , S. , Liu , X. , Mao , H. , Pu , J. , Pedram , A. , Horowitz , M.A. , Dally , W.J. EIE : Efficient inference engine on compressed deep neural network. In ISCA ( 2016 ). IEEE , 243--254. Han, S., Liu, X., Mao, H., Pu, J., Pedram, A., Horowitz, M.A., Dally, W.J. EIE: Efficient inference engine on compressed deep neural network. In ISCA (2016). IEEE, 243--254."},{"key":"e_1_2_1_17_1","volume-title":"ICLR","author":"Han S.","year":"2016","unstructured":"Han , S. , Mao , H. , Dally , W.J. Deep compression: Compressing deep neural networks with pruning, trained quantization and Huffman coding . In ICLR ( 2016 ). Han, S., Mao, H., Dally, W.J. Deep compression: Compressing deep neural networks with pruning, trained quantization and Huffman coding. In ICLR (2016)."},{"key":"e_1_2_1_18_1","volume-title":"NIPS","author":"Han S.","year":"2015","unstructured":"Han , S. , Pool , J. , Tran , J. , Dally , W. Learning both weights and connections for efficient neural network . In NIPS ( 2015 ), 1135--1143. Han, S., Pool, J., Tran, J., Dally, W. Learning both weights and connections for efficient neural network. In NIPS (2015), 1135--1143."},{"key":"e_1_2_1_20_1","volume-title":"ASPDAC","author":"Hartenstein R.","year":"2001","unstructured":"Hartenstein , R. Coarse grain re-configurable architecture (Embedded Tutorial) . In ASPDAC ( 2001 ), ACM , 564--570. Hartenstein, R. Coarse grain re-configurable architecture (Embedded Tutorial). In ASPDAC (2001), ACM, 564--570."},{"key":"e_1_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"e_1_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1145\/3282307"},{"key":"e_1_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1109\/MCSE.2018.021651341"},{"key":"e_1_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1109\/ISSCC.2014.6757323"},{"key":"e_1_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1038\/nbt.4060"},{"key":"e_1_2_1_26_1","volume-title":"Dissecting the NVIDIA turing T4 GPU via microbenchmarking. arXiv:1903.07486","author":"Jia Z.","year":"2019","unstructured":"Jia , Z. , Maggioni , M. , Smith , J. , Scarpazza , D.P. Dissecting the NVIDIA turing T4 GPU via microbenchmarking. arXiv:1903.07486 ( 2019 ). Jia, Z., Maggioni, M., Smith, J., Scarpazza, D.P. Dissecting the NVIDIA turing T4 GPU via microbenchmarking. arXiv:1903.07486 (2019)."},{"key":"e_1_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1145\/3154484"},{"key":"e_1_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1145\/3093337.3037749"},{"key":"e_1_2_1_29_1","first-page":"2","article-title":"Measuring the gap between FPGAs and ASICs","volume":"26","author":"Kuon I.","year":"2007","unstructured":"Kuon , I. , Rose , J . Measuring the gap between FPGAs and ASICs . IEEE TCAD 26 , 2 ( 2007 ), 203--215. Kuon, I., Rose, J. Measuring the gap between FPGAs and ASICs. IEEE TCAD 26, 2 (2007), 203--215.","journal-title":"IEEE TCAD"},{"key":"e_1_2_1_30_1","volume-title":"Comparing Long Strings on a Short Systolic Array","author":"Lipton R.J.","year":"1986","unstructured":"Lipton , R.J. , Lopresti , D.P. Comparing Long Strings on a Short Systolic Array . Princeton University , Department of Computer Science, 1986 . Lipton, R.J., Lopresti, D.P. Comparing Long Strings on a Short Systolic Array. Princeton University, Department of Computer Science, 1986."},{"key":"e_1_2_1_31_1","volume-title":"System power calculators","author":"MICRON.","year":"2019","unstructured":"MICRON. System power calculators , 2019 . https:\/\/tinyurl.com\/y5cvl857 MICRON. System power calculators, 2019. https:\/\/tinyurl.com\/y5cvl857"},{"key":"e_1_2_1_32_1","unstructured":"Moore G.E. etal Cramming more components onto integrated circuits. 1965  Moore G.E. et al. Cramming more components onto integrated circuits. 1965"},{"key":"e_1_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1109\/MM.2010.41"},{"key":"e_1_2_1_34_1","first-page":"2","article-title":"The J-machine multicomputer: An architectural evaluation","volume":"21","author":"Noakes M.D.","year":"1993","unstructured":"Noakes , M.D. , Wallach , D.A. , Dally , W.J . The J-machine multicomputer: An architectural evaluation . Comput. Arch. News 21 , 2 ( 1993 ), 224--235. Noakes, M.D., Wallach, D.A., Dally, W.J. The J-machine multicomputer: An architectural evaluation. Comput. Arch. News 21, 2 (1993), 224--235.","journal-title":"Comput. Arch. News"},{"key":"e_1_2_1_35_1","volume-title":"NVIDIA deep learning accelerator (NVDLA)","author":"NVIDIA.","year":"2017","unstructured":"NVIDIA. NVIDIA deep learning accelerator (NVDLA) , 2017 . http:\/\/nvdla.org NVIDIA. NVIDIA deep learning accelerator (NVDLA), 2017. http:\/\/nvdla.org"},{"key":"e_1_2_1_36_1","volume-title":"NVIDIA Tesla deep learning product performance","author":"NVIDIA.","year":"2019","unstructured":"NVIDIA. NVIDIA Tesla deep learning product performance , 2019 . https:\/\/tinyurl.com\/yyu9amxh NVIDIA. NVIDIA Tesla deep learning product performance, 2019. https:\/\/tinyurl.com\/yyu9amxh"},{"key":"e_1_2_1_37_1","volume-title":"ISCA","author":"Parashar A.","year":"2017","unstructured":"Parashar , A. , SCNN: An accelerator for compressed-sparse convolutional neural networks . In ISCA ( 2017 ), IEEE , 27--40. Parashar, A., et al. SCNN: An accelerator for compressed-sparse convolutional neural networks. In ISCA (2017), IEEE, 27--40."},{"key":"e_1_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1145\/2508148.2485925"},{"key":"e_1_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1145\/2499370.2462176"},{"key":"e_1_2_1_40_1","volume-title":"Handbook of Theoretical Computer Science. North-Holland","author":"Karpand R.M.","year":"1988","unstructured":"Karpand , R.M. , Ramachandran , V. , Karpand , V. , Karp , R.M. A survey of parallel algorithms for shared-memory machines . In Handbook of Theoretical Computer Science. North-Holland , 1988 Karpand, R.M., Ramachandran, V., Karpand, V., Karp, R.M. A survey of parallel algorithms for shared-memory machines. In Handbook of Theoretical Computer Science. North-Holland, 1988"},{"key":"e_1_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1145\/342001.339668"},{"key":"e_1_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1145\/248209.237144"},{"key":"e_1_2_1_43_1","volume-title":"Intel CPU outperforms NVIDIA GPU on ResNet-50 deep learning inference","author":"Shen H.","year":"2019","unstructured":"Shen , H. , Intel CPU outperforms NVIDIA GPU on ResNet-50 deep learning inference , 2019 . https:\/\/tinyurl.com\/y6xewz8r Shen, H., et al. Intel CPU outperforms NVIDIA GPU on ResNet-50 deep learning inference, 2019. https:\/\/tinyurl.com\/y6xewz8r"},{"key":"e_1_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1016\/0022-2836(81)90087-5"},{"key":"e_1_2_1_45_1","first-page":"7","article-title":"Fast and sensitive mapping of nanopore sequencing reads with GraphMap","author":"Sovi\u0107 I.","year":"2016","unstructured":"Sovi\u0107 , I. , \u0160iki\u0107 , M. , Wilm , A. , Fenlon , S.N. , Chen , S. , Nagarajan , N . Fast and sensitive mapping of nanopore sequencing reads with GraphMap . Nat. Commun. 7 , ( 2016 ). Sovi\u0107, I., \u0160iki\u0107, M., Wilm, A., Fenlon, S.N., Chen, S., Nagarajan, N. Fast and sensitive mapping of nanopore sequencing reads with GraphMap. Nat. Commun. 7, (2016).","journal-title":"Nat. Commun."},{"key":"e_1_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.1109\/MCSE.2010.69"},{"key":"e_1_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.1109\/LCA.2019.2909867"},{"key":"e_1_2_1_48_1","doi-asserted-by":"publisher","DOI":"10.5555\/1502144"},{"key":"e_1_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.1145\/3173162.3173193"},{"key":"e_1_2_1_50_1","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA.2019.00050"},{"key":"e_1_2_1_52_1","doi-asserted-by":"publisher","DOI":"10.5555\/645818.669220"},{"key":"e_1_2_1_53_1","volume-title":"Xilinx Imagenet Benchmarks","author":"Xilinx","year":"2019","unstructured":"Xilinx . Xilinx Imagenet Benchmarks , 2019 . https:\/\/tinyurl.com\/y5l4ajff Xilinx. Xilinx Imagenet Benchmarks, 2019. https:\/\/tinyurl.com\/y5l4ajff"}],"container-title":["Communications of the ACM"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3361682","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3361682","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T13:49:27Z","timestamp":1750254567000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3361682"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2020,6,18]]},"references-count":51,"journal-issue":{"issue":"7","published-print":{"date-parts":[[2020,6,18]]}},"alternative-id":["10.1145\/3361682"],"URL":"https:\/\/doi.org\/10.1145\/3361682","relation":{},"ISSN":["0001-0782","1557-7317"],"issn-type":[{"value":"0001-0782","type":"print"},{"value":"1557-7317","type":"electronic"}],"subject":[],"published":{"date-parts":[[2020,6,18]]},"assertion":[{"value":"2020-06-18","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}