{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,4]],"date-time":"2025-12-04T09:58:49Z","timestamp":1764842329036,"version":"3.37.3"},"reference-count":30,"publisher":"Springer Science and Business Media LLC","issue":"3-4","license":[{"start":{"date-parts":[[2019,12,1]],"date-time":"2019-12-01T00:00:00Z","timestamp":1575158400000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2019,12,1]],"date-time":"2019-12-01T00:00:00Z","timestamp":1575158400000},"content-version":"vor","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"funder":[{"name":"National Key Research and Development Plan of China","award":["2017YFC0803401"],"award-info":[{"award-number":["2017YFC0803401"]}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["61872335","61732018"],"award-info":[{"award-number":["61872335","61732018"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"International Partnership Program of Chinese Academy of Sciences","award":["171111KYSB20170032"],"award-info":[{"award-number":["171111KYSB20170032"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["CCF Trans. HPC"],"published-print":{"date-parts":[[2019,12]]},"DOI":"10.1007\/s42514-019-00015-7","type":"journal-article","created":{"date-parts":[[2019,12,4]],"date-time":"2019-12-04T12:03:03Z","timestamp":1575460983000},"page":"177-195","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":3,"title":["Applying CNN on a scientific application accelerator based on dataflow architecture"],"prefix":"10.1007","volume":"1","author":[{"given":"Xiaochun","family":"Ye","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Taoran","family":"Xiang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xu","family":"Tan","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yujing","family":"Feng","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Haibin","family":"Wu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Meng","family":"Wu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Dongrui","family":"Fan","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2019,12,4]]},"reference":[{"key":"15_CR1","doi-asserted-by":"publisher","unstructured":"Albericio, J., Judd, P., Hetherington, T., et\u00a0al.: Cnvlutin: ineffectual-neuron-free deep neural network computing. In: 2016 ACM\/IEEE 43rd Annual International Symposium on Computer Architecture (ISCA), pp. 1\u201313 (2016). https:\/\/doi.org\/10.1109\/ISCA.2016.11","DOI":"10.1109\/ISCA.2016.11"},{"key":"15_CR2","unstructured":"Chellapilla, K., Puri, S., Simard, P.: High performance convolutional neural networks for document processing. In: Tenth International Workshop on Frontiers in Handwriting Recognition, pp. 1\u20136 (2006)"},{"key":"15_CR3","doi-asserted-by":"publisher","unstructured":"Chen, T., Du, Z., Sun, N., et\u00a0al.: DianNao: a small-footprint high-throughput accelerator for ubiquitous machine-learning. In: Proceedings of the 19th International Conference on Architectural Support for Programming Languages and Operating Systems, ASPLOS \u201914, pp. 269\u2013284. ACM, New York (2014). https:\/\/doi.org\/10.1145\/2541940.2541967","DOI":"10.1145\/2541940.2541967"},{"key":"15_CR4","doi-asserted-by":"publisher","unstructured":"Chen, Y., Luo, T., Liu, S., et\u00a0al.: Dadiannao: a machine-learning supercomputer. In: 2014 47th Annual IEEE\/ACM International Symposium on Microarchitecture, pp. 609\u2013622 (2014). https:\/\/doi.org\/10.1109\/MICRO.2014.58","DOI":"10.1109\/MICRO.2014.58"},{"key":"15_CR5","doi-asserted-by":"publisher","unstructured":"Chen, Y.H., Emer, J., Sze, V.: Eyeriss: A spatial architecture for energy-efficient dataflow for convolutional neural networks. In: 2016 ACM\/IEEE 43rd Annual International Symposium on Computer Architecture (ISCA), pp. 367\u2013379 (2016). https:\/\/doi.org\/10.1109\/ISCA.2016.40","DOI":"10.1109\/ISCA.2016.40"},{"key":"15_CR6","unstructured":"Chetlur, S., Woolley, C., Vandermersch, P., et\u00a0al.: cuDNN: efficient primitives for deep learning. CoRR arxiv: abs\/1410.0759 (2014)"},{"key":"15_CR7","unstructured":"Coates, A., Huval, B., Wang, T., et\u00a0al.: Deep learning with COTS HPC systems. In: Proceedings of the 30th International Conference on Machine Learning, vol. 28. ICML\u201913, pp. III-1337\u2013III-1345. JMLR.org (2013). http:\/\/dl.acm.org\/citation.cfm?id=3042817.3043086"},{"key":"15_CR8","first-page":"362","volume-title":"Lecture Notes in Computer Science","author":"Jack B. Dennis","year":"1974","unstructured":"Dennis, J.B.: First version of a data flow procedure language. In: Programming Symposium, Proceedings Colloque Sur La Programmation, pp. 362\u2013376. Springer, London (1974). http:\/\/dl.acm.org\/citation.cfm?id=647323.721501"},{"issue":"2","key":"15_CR9","doi-asserted-by":"publisher","first-page":"38","DOI":"10.1109\/MM.2012.32","volume":"32","author":"D Fan","year":"2012","unstructured":"Fan, D., Zhang, H., Wang, D., Ye, X., Song, F., Li, G., Sun, N.: Godson-T: an efficient many-core processor exploring thread-level parallelism. IEEE Micro 32(2), 38\u201347 (2012)","journal-title":"IEEE Micro"},{"key":"15_CR10","doi-asserted-by":"crossref","unstructured":"Fan, D., Li, W., Ye, X., Wang, D., Zhang, H., Tang, Z., Sun, N.: SmarCO: an efficient many-core processor for high-throughput applications in datacenters. In: 2018 IEEE International Symposium on High Performance Computer Architecture (HPCA), pp. 596\u2013607. IEEE, New York (2018)","DOI":"10.1109\/HPCA.2018.00057"},{"issue":"1","key":"15_CR11","doi-asserted-by":"publisher","first-page":"30","DOI":"10.1109\/MM.2013.111","volume":"34","author":"H Fu","year":"2014","unstructured":"Fu, H., Gan, L., Clapp, R.G., Ruan, H., Pell, O., Mencer, O., Flynn, M., Huang, X., Yang, G.: Scaling reverse time migration performance through reconfigurable dataflow engines. IEEE Micro 34(1), 30\u201340 (2014)","journal-title":"IEEE Micro"},{"key":"15_CR12","doi-asserted-by":"publisher","unstructured":"Govindan, M.S.S., Burger, D., Keckler, S.: Trips: a distributed explicit data graph execution (edge) microprocessor. In: 2007 IEEE Hot Chips 19 Symposium (HCS), pp. 1\u201313 (2007). https:\/\/doi.org\/10.1109\/HOTCHIPS.2007.7482519","DOI":"10.1109\/HOTCHIPS.2007.7482519"},{"key":"15_CR13","doi-asserted-by":"publisher","unstructured":"Gu, L., Li, X., Siegel, J.: An empirically tuned 2D and 3D FFT library on CUDA GPU. In: Proceedings of the 24th ACM International Conference on Supercomputing, ICS \u201910, pp. 305\u2013314. ACM, New York (2010). https:\/\/doi.org\/10.1145\/1810085.1810127","DOI":"10.1145\/1810085.1810127"},{"key":"15_CR14","doi-asserted-by":"publisher","unstructured":"Jouppi, N.P., Young, C., Patil, N., et\u00a0al.: In-datacenter performance analysis of a tensor processing unit. In: Proceedings of the 44th Annual International Symposium on Computer Architecture, ISCA \u201917, pp. 1\u201312. ACM, New York (2017). https:\/\/doi.org\/10.1145\/3079856.3080246","DOI":"10.1145\/3079856.3080246"},{"key":"15_CR15","unstructured":"Krizhevsky, A., Sutskever, I., Hinton, G.E.: ImageNet classification with deep convolutional neural networks. In: F.\u00a0Pereira, C.J.C. Burges, L.\u00a0Bottou, K.Q. Weinberger (eds.) Advances in Neural Information Processing Systems, vol. 25, pp. 1097\u20131105. Curran Associates, Inc., Red Hook (2012)"},{"key":"15_CR16","doi-asserted-by":"publisher","unstructured":"Liang, Y., Lu, L., Xiao, Q., Yan, S.: Evaluating fast algorithms for convolutional neural networks on FPGAS. In: IEEE Transactions on Computer-Aided Design of Integrated Circuits and Systems, pp. 1\u20131 (2019). https:\/\/doi.org\/10.1109\/TCAD.2019.2897701","DOI":"10.1109\/TCAD.2019.2897701"},{"key":"15_CR17","doi-asserted-by":"publisher","unstructured":"Lu, L., Liang, Y.: SpWA: an efficient sparse Winograd convolutional neural networks accelerator on FPGAS. In: 2018 55th ACM\/ESDA\/IEEE Design Automation Conference (DAC), pp. 1\u20136 (2018). https:\/\/doi.org\/10.1109\/DAC.2018.8465842","DOI":"10.1109\/DAC.2018.8465842"},{"key":"15_CR18","doi-asserted-by":"publisher","unstructured":"Lu, W., Yan, G., Li, J., et\u00a0al.: FlexFlow: a flexible dataflow accelerator architecture for convolutional neural networks. In: 2017 IEEE International Symposium on High Performance Computer Architecture (HPCA), pp. 553\u2013564 (2017). https:\/\/doi.org\/10.1109\/HPCA.2017.29","DOI":"10.1109\/HPCA.2017.29"},{"key":"15_CR19","doi-asserted-by":"publisher","unstructured":"Nguyen, A., Satish, N., Chhugani, J., et\u00a0al.: 3.5-D blocking optimization for stencil computations on modern CPUs and GPUs. In: 2010 ACM\/IEEE International Conference for High Performance Computing, Networking, Storage and Analysis, pp. 1\u201313 (2010). https:\/\/doi.org\/10.1109\/SC.2010.2","DOI":"10.1109\/SC.2010.2"},{"key":"15_CR20","doi-asserted-by":"crossref","unstructured":"Oriato, D., Tilbury, S., Marrocu, M., Pusceddu, G.: Acceleration of a meteorological limited area model with dataflow engines. In: 2012 Symposium on Application Accelerators in High Performance Computing (SAAHPC), pp. 129\u2013132. IEEE, New York (2012)","DOI":"10.1109\/SAAHPC.2012.8"},{"key":"15_CR21","doi-asserted-by":"crossref","unstructured":"Pratas, F., Oriato, D., Pell, O., Mata, R.A., Sousa, L.: Accelerating the computation of induced dipoles for molecular mechanics with dataflow engines. In: IEEE 21st Annual International Symposium on Field-Programmable Custom Computing Machines, pp. 177\u2013180. IEEE, New York (2013)","DOI":"10.1109\/FCCM.2013.34"},{"key":"15_CR22","unstructured":"Simonyan, K., Zisserman, A.: Very deep convolutional networks for large-scale image recognition. arXiv preprint arXiv:1409.1556 (2014)"},{"issue":"2","key":"15_CR23","doi-asserted-by":"publisher","first-page":"4:1","DOI":"10.1145\/1233307.1233308","volume":"25","author":"S Swanson","year":"2007","unstructured":"Swanson, S., Schwerin, A., Mercaldi, M., et al.: The wavescalar architecture. ACM Trans. Comput. Syst. 25(2), 4:1\u20134:54 (2007). https:\/\/doi.org\/10.1145\/1233307.1233308","journal-title":"ACM Trans. Comput. Syst."},{"key":"15_CR24","unstructured":"Sze, V., Chen, Y., Yang, T., Emer, J.S.: Efficient processing of deep neural networks: a tutorial and survey. CoRR arxiv: abs\/1703.09039 (2017)"},{"issue":"1","key":"15_CR25","doi-asserted-by":"publisher","first-page":"116","DOI":"10.1007\/s11390-017-1748-5","volume":"33","author":"X Tan","year":"2018","unstructured":"Tan, X., Ye, X.C., Shen, X.W., Xu, Y.C., Wang, D., Zhang, L., Li, W.M., Fan, D.R., Tang, Z.M.: A pipelining loop optimization method for dataflow architecture. J. Comput. Sci. Technol. 33(1), 116\u2013130 (2018). https:\/\/doi.org\/10.1007\/s11390-017-1748-5","journal-title":"J. Comput. Sci. Technol."},{"key":"15_CR26","unstructured":"Venkataramanan, G., Sarma, D.D.: Compute and redundancy solution for Tesla\u2019s full self driving computer. In: Hotchips 2019 (2019)"},{"key":"15_CR27","doi-asserted-by":"publisher","unstructured":"Verdoscia, L., Vaccaro, R., Giorgi, R.: A matrix multiplier case study for an evaluation of a configurable dataflow-machine. In: Proceedings of the 12th ACM International Conference on Computing Frontiers, CF \u201915, pp. 63:1\u201363:6. ACM, New York (2015). https:\/\/doi.org\/10.1145\/2742854.2747287","DOI":"10.1145\/2742854.2747287"},{"key":"15_CR28","first-page":"2181","volume":"9","author":"S Xiao-Wei","year":"2017","unstructured":"Xiao-Wei, S., Xiao-Chun, Y., Da, W., et al.: Optimizing dataflow architecture for scientific applications. Chin. J. Comput. 9, 2181\u20132196 (2017)","journal-title":"Chin. J. Comput."},{"key":"15_CR29","doi-asserted-by":"publisher","unstructured":"Ye, X., Fan, D., Sun, N., et\u00a0al.: SimICT: a fast and flexible framework for performance and power evaluation of large-scale architecture. In: International Symposium on Low Power Electronics and Design (ISLPED), pp. 273\u2013278 (2013). https:\/\/doi.org\/10.1109\/ISLPED.2013.6629308","DOI":"10.1109\/ISLPED.2013.6629308"},{"key":"15_CR30","doi-asserted-by":"publisher","unstructured":"Zhang, C., Li, P., Sun, G., et\u00a0al.: Optimizing FPGA-based accelerator design for deep convolutional neural networks. In: Proceedings of the 2015 ACM\/SIGDA International Symposium on Field-Programmable Gate Arrays, FPGA \u201915, pp. 161\u2013170. ACM, New York (2015). https:\/\/doi.org\/10.1145\/2684746.2689060","DOI":"10.1145\/2684746.2689060"}],"container-title":["CCF Transactions on High Performance Computing"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s42514-019-00015-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s42514-019-00015-7\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s42514-019-00015-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2020,12,3]],"date-time":"2020-12-03T03:06:04Z","timestamp":1606964764000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s42514-019-00015-7"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2019,12]]},"references-count":30,"journal-issue":{"issue":"3-4","published-print":{"date-parts":[[2019,12]]}},"alternative-id":["15"],"URL":"https:\/\/doi.org\/10.1007\/s42514-019-00015-7","relation":{},"ISSN":["2524-4922","2524-4930"],"issn-type":[{"type":"print","value":"2524-4922"},{"type":"electronic","value":"2524-4930"}],"subject":[],"published":{"date-parts":[[2019,12]]},"assertion":[{"value":"31 May 2019","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"25 October 2019","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"4 December 2019","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}