{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,3,27]],"date-time":"2025-03-27T17:56:56Z","timestamp":1743098216383,"version":"3.40.3"},"publisher-location":"Singapore","reference-count":27,"publisher":"Springer Nature Singapore","isbn-type":[{"type":"print","value":"9789819615445"},{"type":"electronic","value":"9789819615452"}],"license":[{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-981-96-1545-2_8","type":"book-chapter","created":{"date-parts":[[2025,2,12]],"date-time":"2025-02-12T16:55:11Z","timestamp":1739379311000},"page":"114-133","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Compression Format and\u00a0Systolic Array Structure Co-design for\u00a0Accelerating Sparse Matrix Multiplication in\u00a0DNNs"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0009-0000-2020-5116","authenticated-orcid":false,"given":"Yongxiang","family":"Cao","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jixiang","family":"Jiang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-5477-3020","authenticated-orcid":false,"given":"Guocheng","family":"Zhao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-4416-2948","authenticated-orcid":false,"given":"Wei","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6633-6605","authenticated-orcid":false,"given":"Hongxu","family":"Jiang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9218-1018","authenticated-orcid":false,"given":"Yanfei","family":"Song","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,2,13]]},"reference":[{"key":"8_CR1","doi-asserted-by":"publisher","unstructured":"Blaiech, A.G., Khalifa, K.B., Valderrama, C., Fernandes, M.A.C., Bedoui, M.H.: A survey and taxonomy of fpga-based deep learning accelerators. J. Syst. Archit. 98, 331\u2013345 (2019). https:\/\/doi.org\/10.1016\/J.SYSARC.2019.01.007","DOI":"10.1016\/J.SYSARC.2019.01.007"},{"key":"8_CR2","doi-asserted-by":"publisher","unstructured":"Chen, X., Zhu, J., Jiang, J., Tsui, C.: Tight compression: compressing CNN through fine-grained pruning and weight permutation for efficient implementation. IEEE Trans. Comput. Aided Des. Integr. Circuits Syst. 42(2), 644\u2013657 (2023). https:\/\/doi.org\/10.1109\/TCAD.2022.3178047","DOI":"10.1109\/TCAD.2022.3178047"},{"key":"8_CR3","unstructured":"Cho, M., Brand, D.: MEC: memory-efficient convolution for deep neural network. In: Precup, D., Teh, Y.W. (eds.) Proceedings of the 34th International Conference on Machine Learning, ICML 2017, Sydney, NSW, Australia, 6-11 August 2017. Proceedings of Machine Learning Research, vol.\u00a070, pp. 815\u2013824. PMLR (2017). http:\/\/proceedings.mlr.press\/v70\/cho17a.html"},{"key":"8_CR4","doi-asserted-by":"publisher","unstructured":"Das, S., Roy, A., Chandrasekharan, K.K., Deshwal, A., Lee, S.: A systolic dataflow based accelerator for CNNs. In: IEEE International Symposium on Circuits and Systems, ISCAS 2020, Sevilla, Spain, October 10-21, 2020, pp.\u00a01\u20135. IEEE (2020). https:\/\/doi.org\/10.1109\/ISCAS45731.2020.9180403","DOI":"10.1109\/ISCAS45731.2020.9180403"},{"key":"8_CR5","doi-asserted-by":"publisher","unstructured":"Genc, H., et al.: Gemmini: enabling systematic deep-learning architecture evaluation via full-stack integration. In: 58th ACM\/IEEE Design Automation Conference, DAC 2021, San Francisco, CA, USA, 5\u20139 December, 2021, pp. 769\u2013774. IEEE (2021). https:\/\/doi.org\/10.1109\/DAC18074.2021.9586216","DOI":"10.1109\/DAC18074.2021.9586216"},{"key":"8_CR6","doi-asserted-by":"publisher","unstructured":"Gupta, M., Agrawal, P.: Compression of deep learning models for text: a survey. ACM Trans. Knowl. Discov. Data 16(4), 61:1\u201361:55 (2022). https:\/\/doi.org\/10.1145\/3487045","DOI":"10.1145\/3487045"},{"key":"8_CR7","doi-asserted-by":"publisher","unstructured":"Han, S., et al.: Deep compression and EIE: efficient inference engine on compressed deep neural network. In: 2016 IEEE Hot Chips 28 Symposium (HCS), Cupertino, CA, USA, 21\u201323 August, 2016, pp.\u00a01\u20136. IEEE (2016). https:\/\/doi.org\/10.1109\/HOTCHIPS.2016.7936226","DOI":"10.1109\/HOTCHIPS.2016.7936226"},{"key":"8_CR8","doi-asserted-by":"publisher","unstructured":"He, X., et al.: Sparse-tpu: adapting systolic arrays for sparse matrices. In: Ayguad\u00e9, E., Hwu, W.W., Badia, R.M., Hofstee, H.P. (eds.) ICS \u201920: 2020 International Conference on Supercomputing, Barcelona Spain, June, 2020, pp. 19:1\u201319:12. ACM (2020). https:\/\/doi.org\/10.1145\/3392717.3392751","DOI":"10.1145\/3392717.3392751"},{"key":"8_CR9","doi-asserted-by":"publisher","unstructured":"Karakasis, V., Gkountouvas, T., Kourtis, K., Goumas, G., Koziris, N.: An extended compression format for the optimization of sparse matrix-vector multiplication. IEEE Trans. Parallel Distrib. Syst. 24(10), 1930\u20131940 (2013). https:\/\/doi.org\/10.1109\/TPDS.2012.290","DOI":"10.1109\/TPDS.2012.290"},{"issue":"1","key":"8_CR10","doi-asserted-by":"publisher","first-page":"91","DOI":"10.1177\/1094342015597082","volume":"31","author":"S Kirmani","year":"2017","unstructured":"Kirmani, S., Park, J., Raghavan, P.: An embedded sectioning scheme for multiprocessor topology-aware mapping of irregular applications. Int. J. High Performance Comput. Appl. 31(1), 91\u2013103 (2017)","journal-title":"Int. J. High Performance Comput. Appl."},{"key":"8_CR11","doi-asserted-by":"publisher","unstructured":"Latotzke, C., Ciesielski, T., Gemmeke, T.: Design of high-throughput mixed-precision CNN accelerators on FPGA. In: 32nd International Conference on Field-Programmable Logic and Applications, FPL 2022, Belfast, United Kingdom, August 29\u2013Sept. 2, 2022, pp. 358\u2013365. IEEE (2022). https:\/\/doi.org\/10.1109\/FPL57034.2022.00061","DOI":"10.1109\/FPL57034.2022.00061"},{"key":"8_CR12","doi-asserted-by":"publisher","unstructured":"Liu, Z.G., Whatmough, P.N., Mattina, M.: Systolic tensor array: an efficient structured-sparse GEMM accelerator for mobile CNN inference. IEEE Comput. Archit. Lett. 19(1), 34\u201337 (2020). https:\/\/doi.org\/10.1109\/LCA.2020.2979965","DOI":"10.1109\/LCA.2020.2979965"},{"key":"8_CR13","doi-asserted-by":"crossref","unstructured":"Lu, L., Xie, J., Huang, R., Zhang, J., Lin, W., Liang, Y.: An efficient hardware accelerator for sparse convolutional neural networks on fpgas. In: 2019 IEEE 27th Annual International Symposium on Field-Programmable Custom Computing Machines (FCCM), pp. 17\u201325. IEEE (2019)","DOI":"10.1109\/FCCM.2019.00013"},{"key":"8_CR14","doi-asserted-by":"publisher","unstructured":"Meng, J., Venkataramanaiah, S.K., Zhou, C., Hansen, P., Whatmough, P.N., Seo, J.: Fixyfpga: Efficient FPGA accelerator for deep neural networks with high element-wise sparsity and without external memory access. In: 31st International Conference on Field-Programmable Logic and Applications, FPL 2021, Dresden, Germany, August 30\u2013Sept. 3, 2021, pp. 9\u201316. IEEE (2021). https:\/\/doi.org\/10.1109\/FPL53798.2021.00010","DOI":"10.1109\/FPL53798.2021.00010"},{"key":"8_CR15","doi-asserted-by":"publisher","unstructured":"Mi, H., Yu, X., Yu, X., Wu, S., Liu, W.: Balancing computation and communication in distributed sparse matrix-vector multiplication. In: Simmhan, Y., Altintas, I., Varbanescu, A.L., Balaji, P., Prasad, A.S., Carnevale, L. (eds.) 23rd IEEE\/ACM International Symposium on Cluster, Cloud and Internet Computing, CCGrid 2023, Bangalore, India, 1\u20134 May, 2023, pp. 535\u2013544. IEEE (2023). https:\/\/doi.org\/10.1109\/CCGRID57682.2023.00056","DOI":"10.1109\/CCGRID57682.2023.00056"},{"key":"8_CR16","doi-asserted-by":"publisher","unstructured":"Parashar, A., et al.: SCNN: an accelerator for compressed-sparse convolutional neural networks. In: Proceedings of the 44th Annual International Symposium on Computer Architecture, ISCA 2017, Toronto, ON, Canada, 24\u201328 June, 2017, pp. 27\u201340. ACM (2017). https:\/\/doi.org\/10.1145\/3079856.3080254","DOI":"10.1145\/3079856.3080254"},{"key":"8_CR17","doi-asserted-by":"publisher","unstructured":"Shomron, G., Horowitz, T., Weiser, U.C.: SMT-SA: simultaneous multithreading in systolic arrays. IEEE Comput. Archit. Lett. 18(2), 99\u2013102 (2019). https:\/\/doi.org\/10.1109\/LCA.2019.2924007","DOI":"10.1109\/LCA.2019.2924007"},{"key":"8_CR18","doi-asserted-by":"publisher","unstructured":"Soltaniyeh, M., Martin, R.P., Nagarakatte, S.: An accelerator for sparse convolutional neural networks leveraging systolic general matrix-matrix multiplication. ACM Trans. Archit. Code Optim. 19(3), 42:1\u201342:26 (2022). https:\/\/doi.org\/10.1145\/3532863","DOI":"10.1145\/3532863"},{"key":"8_CR19","doi-asserted-by":"crossref","unstructured":"Sun, W., Liu, D., Zou, Z., Sun, W., Chen, S., Kang, Y.: Sense: Model-hardware codesign for accelerating sparse CNNs on systolic arrays. IEEE Trans. Very Large Scale Integration (VLSI) Syst. 31(4), 470\u2013483 (2023)","DOI":"10.1109\/TVLSI.2023.3241933"},{"key":"8_CR20","doi-asserted-by":"publisher","unstructured":"Wei, X., et al.: Automated systolic array architecture synthesis for high throughput CNN inference on fpgas. In: Proceedings of the 54th Annual Design Automation Conference, DAC 2017, Austin, TX, USA, June 18-22, 2017, pp. 29:1\u201329:6. ACM (2017). https:\/\/doi.org\/10.1145\/3061639.3062207","DOI":"10.1145\/3061639.3062207"},{"key":"8_CR21","doi-asserted-by":"publisher","unstructured":"Xu, R., Ma, S., Guo, Y., Li, D.: A survey of design and optimization for systolic array-based DNN accelerators. ACM Comput. Surv. 56(1), 20:1\u201320:37 (2024). https:\/\/doi.org\/10.1145\/3604802","DOI":"10.1145\/3604802"},{"key":"8_CR22","doi-asserted-by":"publisher","unstructured":"Yang, J., Fu, W., Cheng, X., Ye, X., Dai, P., Zhao, W.: S$$ ^{\\text{2}}$$ engine: a novel systolic architecture for sparse convolutional neural networks. IEEE Trans. Comput. 71(6), 1440\u20131452 (2022). https:\/\/doi.org\/10.1109\/TC.2021.3087946","DOI":"10.1109\/TC.2021.3087946"},{"key":"8_CR23","doi-asserted-by":"publisher","unstructured":"Yang, Y., Kuppannagari, S.R., Kannan, R., Prasanna, V.K.: Bandwidth efficient homomorphic encrypted matrix vector multiplication accelerator on FPGA. In: International Conference on Field-Programmable Technology, (IC)FPT 2022, Hong Kong, 5\u20139 December, 2022, pp.\u00a01\u20139. IEEE (2022). https:\/\/doi.org\/10.1109\/ICFPT56656.2022.9974369","DOI":"10.1109\/ICFPT56656.2022.9974369"},{"key":"8_CR24","doi-asserted-by":"crossref","unstructured":"Yin, X., Wu, Z., Li, D., Shen, C., Liu, Y.: An efficient hardware accelerator for block sparse convolutional neural networks on fpga. IEEE Embedded Systems Letters (2023)","DOI":"10.1109\/LES.2023.3296507"},{"key":"8_CR25","doi-asserted-by":"publisher","unstructured":"Zhang, J., Gu, H., Zhang, G.L., Li, B., Schlichtmann, U.: Hardware-software codesign of weight reshaping and systolic array multiplexing for efficient CNNs. In: Design, Automation & Test in Europe Conference & Exhibition, DATE 2021, Grenoble, France, 1\u20135 February, 2021, pp. 667\u2013672. IEEE (2021). https:\/\/doi.org\/10.23919\/DATE51398.2021.9474215","DOI":"10.23919\/DATE51398.2021.9474215"},{"key":"8_CR26","doi-asserted-by":"publisher","unstructured":"Zhang, Z., Wang, H., Han, S., Dally, W.J.: Sparch: Efficient architecture for sparse matrix multiplication. In: IEEE International Symposium on High Performance Computer Architecture, HPCA 2020, San Diego, CA, USA, 22\u201326 February, 2020, pp. 261\u2013274. IEEE (2020). https:\/\/doi.org\/10.1109\/HPCA47549.2020.00030, https:\/\/doi.org\/10.1109\/HPCA47549.2020.00030","DOI":"10.1109\/HPCA47549.2020.00030"},{"key":"8_CR27","doi-asserted-by":"publisher","unstructured":"Zhu, C., Huang, K., Yang, S., Zhu, Z., Zhang, H., Shen, H.: An efficient hardware accelerator for structured sparse convolutional neural networks on fpgas. IEEE Trans. Very Large Scale Integr. Syst. 28(9), 1953\u20131965 (2020). https:\/\/doi.org\/10.1109\/TVLSI.2020.3002779","DOI":"10.1109\/TVLSI.2020.3002779"}],"container-title":["Lecture Notes in Computer Science","Algorithms and Architectures for Parallel Processing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-96-1545-2_8","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,2,12]],"date-time":"2025-02-12T16:55:17Z","timestamp":1739379317000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-96-1545-2_8"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025]]},"ISBN":["9789819615445","9789819615452"],"references-count":27,"URL":"https:\/\/doi.org\/10.1007\/978-981-96-1545-2_8","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2025]]},"assertion":[{"value":"13 February 2025","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ICA3PP","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Algorithms and Architectures for Parallel Processing","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Macau","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"China","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"30 October 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"1 November 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"24","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"ica3pp2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/ica3pp2024.scimeeting.cn\/en\/web\/index\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}