{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,9,26]],"date-time":"2025-09-26T13:05:15Z","timestamp":1758891915096},"reference-count":30,"publisher":"Springer Science and Business Media LLC","issue":"4","license":[{"start":{"date-parts":[[2017,10,3]],"date-time":"2017-10-03T00:00:00Z","timestamp":1506988800000},"content-version":"unspecified","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"funder":[{"DOI":"10.13039\/501100001809","name":"NSFC","doi-asserted-by":"crossref","award":["61379040"],"award-info":[{"award-number":["61379040"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"crossref"}]},{"name":"CCF-Venustech Hongyan Research","award":["CCF-VenustechRP1026002"],"award-info":[{"award-number":["CCF-VenustechRP1026002"]}]},{"name":"Anhui Provincial NSF","award":["608085QF12"],"award-info":[{"award-number":["608085QF12"]}]},{"name":"Suzhou Research","award":["SYG201625"],"award-info":[{"award-number":["SYG201625"]}]},{"DOI":"10.13039\/501100004739","name":"Youth Innovation Promotion Association CAS","doi-asserted-by":"crossref","award":["2017497"],"award-info":[{"award-number":["2017497"]}],"id":[{"id":"10.13039\/501100004739","id-type":"DOI","asserted-by":"crossref"}]},{"DOI":"10.13039\/501100012226","name":"Fundamental Research Funds for the Central Universities","doi-asserted-by":"crossref","award":["WK2150110003"],"award-info":[{"award-number":["WK2150110003"]}],"id":[{"id":"10.13039\/501100012226","id-type":"DOI","asserted-by":"crossref"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Parallel Prog"],"published-print":{"date-parts":[[2018,8]]},"DOI":"10.1007\/s10766-017-0528-8","type":"journal-article","created":{"date-parts":[[2017,10,4]],"date-time":"2017-10-04T08:01:10Z","timestamp":1507104070000},"page":"648-659","update-policy":"http:\/\/dx.doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":5,"title":["SparseNN: A Performance-Efficient Accelerator for Large-Scale Sparse Neural Networks"],"prefix":"10.1007","volume":"46","author":[{"given":"Yuntao","family":"Lu","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Chao","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Lei","family":"Gong","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xuehai","family":"Zhou","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2017,10,3]]},"reference":[{"key":"528_CR1","unstructured":"Abadi, M., Agarwal, A., Barham, P., Brevdo, E., Chen, Z., Citro, C., Corrado, G.S., Davis, A., Dean, J., Devin, M., et\u00a0al.: Tensorflow: Large-Scale Machine Learning on Heterogeneous Distributed Systems. arXiv:1603.04467 (2016)"},{"key":"528_CR2","doi-asserted-by":"crossref","unstructured":"Bergstra, J., Breuleux, O., Bastien, F., Lamblin, P., Pascanu, R., Desjardins, G., Turian, J., Warde-Farley, D., Bengio, Y.: Theano: a CPU and GPU math compiler in python. In: EuroSciPy, pp. 1\u20137 (2010)","DOI":"10.25080\/Majora-92bf1922-003"},{"key":"528_CR3","doi-asserted-by":"crossref","unstructured":"Chen, T., Du, Z., Sun, N., Wang, J., Wu, C., Chen, Y., Temam, O.: Diannao: a small-footprint high-throughput accelerator for ubiquitous machine-learning. In: ACM Sigplan Notices, vol. 49, pp. 269\u2013284 (2014)","DOI":"10.1145\/2541940.2541967"},{"key":"528_CR4","unstructured":"Coates, A., Huval, B., Wang, T., Wu, D., Catanzaro, B., Andrew, N.: Deep learning with COTS HPC systems. In: ICML, pp. 1337\u20131345 (2013)"},{"key":"528_CR5","unstructured":"Collobert, R., Bengio, S., Mari\u00e9thoz, J.: Torch: a modular machine learning software library. Tech. rep. (2002)"},{"key":"528_CR6","doi-asserted-by":"crossref","unstructured":"Du, Z., Fasthuber, R., Chen, T., Ienne, P., Li, L., Luo, T., Feng, X., Chen, Y., Temam, O.: Shidiannao: shifting vision processing closer to the sensor. In: SCAN, vol. 43, pp. 92\u2013104 (2015)","DOI":"10.1145\/2749469.2750389"},{"key":"528_CR7","doi-asserted-by":"crossref","unstructured":"Hameed, R., Qadeer, W., Wachs, M., Azizi, O., Solomatnikov, A., Lee, B.C., Richardson, S., Kozyrakis, C., Horowitz, M.: Understanding sources of inefficiency in general-purpose chips. In: SCAN, vol. 38, pp. 37\u201347 (2010)","DOI":"10.1145\/1815961.1815968"},{"key":"528_CR8","unstructured":"Han, S., Mao, H., Dally, W.J.: Deep Compression: Compressing Deep Neural Networks with Pruning, Trained Quantization and Huffman Coding. arXiv:1510.00149 (2015)"},{"key":"528_CR9","doi-asserted-by":"crossref","unstructured":"Han, S., Liu, X., Mao, H., Pu, J., Pedram, A., Horowitz, M.A., Dally, W.J.: EIE: efficient inference engine on compressed deep neural network. In: ISCA, pp. 243\u2013254 (2016)","DOI":"10.1145\/3007787.3001163"},{"key":"528_CR10","unstructured":"Han, S., Pool, J., Tran, J., Dally, W.: Learning both weights and connections for efficient neural network. In: NIPS, pp. 1135\u20131143 (2016)"},{"key":"528_CR11","doi-asserted-by":"crossref","unstructured":"Jia, Y., Shelhamer, E., Donahue, J., Karayev, S., Long, J., Girshick, R., Guadarrama, S., Darrell, T.: Caffe: convolutional architecture for fast feature embedding. In: ICM, pp. 675\u2013678 (2014)","DOI":"10.1145\/2647868.2654889"},{"key":"528_CR12","unstructured":"Krizhevsky, A., Sutskever, I., Hinton, G.E.: Imagenet classification with deep convolutional neural networks. In: NIPS, pp. 1097\u20131105 (2012)"},{"key":"528_CR13","doi-asserted-by":"crossref","unstructured":"Le, Q.V.: Building high-level features using large scale unsupervised learning. In: ICASSP, pp. 8595\u20138598 (2013)","DOI":"10.1109\/ICASSP.2013.6639343"},{"issue":"11","key":"528_CR14","doi-asserted-by":"crossref","first-page":"2278","DOI":"10.1109\/5.726791","volume":"86","author":"Y LeCun","year":"1998","unstructured":"LeCun, Y., Bottou, L., Bengio, Y., Haffner, P.: Gradient-based learning applied to document recognition. Proc. IEEE 86(11), 2278\u20132324 (1998)","journal-title":"Proc. IEEE"},{"key":"528_CR15","doi-asserted-by":"crossref","unstructured":"Lu, Y., Gong, L., Xu, C., Sun, F., Zhang, Y., Wang, C., Zhou, X.: Work-in-Progress: A High-Performance FPGA Accelerator for Sparse Neural Networks (2017)","DOI":"10.1145\/3125501.3125510"},{"key":"528_CR16","doi-asserted-by":"crossref","unstructured":"Luo, T., Liu, S., Li, L., Wang, Y., Zhang, S., Chen, T., Xu, Z., Temam, O., Chen, Y.: Dadiannao: a neural network supercomputer. TC 66(1), 73\u201388 (2017)","DOI":"10.1109\/TC.2016.2574353"},{"key":"528_CR17","doi-asserted-by":"crossref","unstructured":"Mikolov, T., Karafi\u00e1t, M., Burget, L., Cernock\u1ef3, J., Khudanpur, S.: Recurrent neural network based language model. In: Interspeech, vol.\u00a02, p.\u00a03 (2010)","DOI":"10.1109\/ICASSP.2011.5947611"},{"issue":"6583","key":"528_CR18","doi-asserted-by":"crossref","first-page":"607","DOI":"10.1038\/381607a0","volume":"381","author":"BA Olshausen","year":"1996","unstructured":"Olshausen, B.A., Field, D.J.: Emergence of simple-cell receptive field properties by learning a sparse code for natural images. Nature 381(6583), 607 (1996)","journal-title":"Nature"},{"key":"528_CR19","doi-asserted-by":"crossref","unstructured":"Poultney, C., Chopra, S., Cun, Y.L., et\u00a0al.: Efficient learning of sparse representations with an energy-based model. In: ANIPS, pp. 1137\u20131144 (2007)","DOI":"10.7551\/mitpress\/7503.003.0147"},{"key":"528_CR20","doi-asserted-by":"crossref","unstructured":"Qiu, J., Wang, J., Yao, S., Guo, K., Li, B., Zhou, E., Yu, J., Tang, T., Xu, N., Song, S., et\u00a0al.: Going deeper with embedded FPGA platform for convolutional neural network. In: FPGA, pp. 26\u201335 (2016)","DOI":"10.1145\/2847263.2847265"},{"issue":"1","key":"528_CR21","first-page":"24","volume":"26","author":"A Rafique","year":"2015","unstructured":"Rafique, A., Constantinides, G.A., Kapre, N.: Communication optimization of iterative sparse matrix-vector multiply on GPUs and FPGAs. TPDS 26(1), 24\u201334 (2015)","journal-title":"TPDS"},{"key":"528_CR22","unstructured":"Simonyan, K., Zisserman, A.: Very Deep Convolutional Networks for Large-Scale Image Recognition. arXiv:1409.1556 (2015)"},{"key":"528_CR23","doi-asserted-by":"crossref","unstructured":"Temam, O.: A defect-tolerant accelerator for emerging high-performance applications. In: ISCA, pp. 356\u2013367 (2012)","DOI":"10.1109\/ISCA.2012.6237031"},{"key":"528_CR24","doi-asserted-by":"crossref","unstructured":"Wang, C., Zhang, J., Zhou, X., Feng, X., Wang, A.: A flexible high speed star network based on peer to peer links on FPGA. In: 2011 IEEE 9th International Symposium on Parallel and Distributed Processing with Applications (ISPA), IEEE, pp. 107\u2013112 (2011)","DOI":"10.1109\/ISPA.2011.40"},{"key":"528_CR25","doi-asserted-by":"crossref","unstructured":"Wang, C., Li, X., Chen, P., Zhang, J., Feng, X., Zhou, X.: Regarding processors and reconfigurable IP cores as services. In: 2012 IEEE Ninth International Conference on Services Computing (SCC), pp. 668\u2013669. IEEE (2012)","DOI":"10.1109\/SCC.2012.72"},{"issue":"3","key":"528_CR26","first-page":"513","volume":"36","author":"C Wang","year":"2017","unstructured":"Wang, C., Gong, L., Yu, Q., Li, X., Xie, Y., Zhou, X.: Dlau: a scalable deep learning accelerator unit on fpga. IEEE Trans. Comput. Aided Des. Integr. Circuits Syst. 36(3), 513\u2013517 (2017)","journal-title":"IEEE Trans. Comput. Aided Des. Integr. Circuits Syst."},{"issue":"2","key":"528_CR27","first-page":"113","volume":"3","author":"FY Wang","year":"2016","unstructured":"Wang, F.Y., Zhang, J.J., Zheng, X., Wang, X., Yuan, Y., Dai, X., Zhang, J., Yang, L.: Where does alphago go: from church-turing thesis to alphago thesis and beyond. JAS 3(2), 113\u2013120 (2016)","journal-title":"JAS"},{"key":"528_CR28","doi-asserted-by":"crossref","unstructured":"Yu, Q., Wang, C., Ma, X., Li, X., Zhou, X.: A deep learning prediction process accelerator based FPGA. In: CCGrid, IEEE, pp. 1159\u20131162 (2015)","DOI":"10.1109\/CCGrid.2015.114"},{"key":"528_CR29","doi-asserted-by":"crossref","unstructured":"Zhang, C., Li, P., Sun, G., Guan, Y., Xiao, B., Cong, J.: Optimizing FPGA-based accelerator design for deep convolutional neural networks. In: FPGA, pp. 161\u2013170 (2015)","DOI":"10.1145\/2684746.2689060"},{"key":"528_CR30","doi-asserted-by":"crossref","unstructured":"Zhang, S., Du, Z., Zhang, L., Lan, H., Liu, S., Li, L., Guo, Q., Chen, T., Chen, Y.: Cambricon-x: an accelerator for sparse neural networks. In: MICRO, pp. 1\u201312 (2016)","DOI":"10.1109\/MICRO.2016.7783723"}],"container-title":["International Journal of Parallel Programming"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s10766-017-0528-8\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s10766-017-0528-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s10766-017-0528-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,8,26]],"date-time":"2023-08-26T18:52:42Z","timestamp":1693075962000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s10766-017-0528-8"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2017,10,3]]},"references-count":30,"journal-issue":{"issue":"4","published-print":{"date-parts":[[2018,8]]}},"alternative-id":["528"],"URL":"https:\/\/doi.org\/10.1007\/s10766-017-0528-8","relation":{},"ISSN":["0885-7458","1573-7640"],"issn-type":[{"value":"0885-7458","type":"print"},{"value":"1573-7640","type":"electronic"}],"subject":[],"published":{"date-parts":[[2017,10,3]]}}}