{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,21]],"date-time":"2026-08-21T13:03:31Z","timestamp":1787317411929,"version":"build-2736575974"},"publisher-location":"New York, NY, USA","reference-count":91,"publisher":"ACM","license":[{"start":{"date-parts":[[2025,2,27]],"date-time":"2025-02-27T00:00:00Z","timestamp":1740614400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"funder":[{"name":"ONR Minerva program"},{"name":"NSF","award":["2107085"],"award-info":[{"award-number":["2107085"]}]},{"name":"iMAGiNE ? the Intelligent Machine Engineering Consortium at UT Austin"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,2,27]]},"DOI":"10.1145\/3706628.3708867","type":"proceedings-article","created":{"date-parts":[[2025,2,26]],"date-time":"2025-02-26T12:22:11Z","timestamp":1740572531000},"page":"159-171","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":11,"title":["Systolic Sparse Tensor Slices: FPGA Building Blocks for Sparse and Dense AI Acceleration"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0000-5136-7580","authenticated-orcid":false,"given":"Endri","family":"Taka","sequence":"first","affiliation":[{"name":"The University of Texas at Austin, Austin, TX, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4663-9099","authenticated-orcid":false,"given":"Ning-Chi","family":"Huang","sequence":"additional","affiliation":[{"name":"National Yang Ming Chiao Tung University, Hsinchu, Taiwan"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-4011-0867","authenticated-orcid":false,"given":"Chi-Chih","family":"Chang","sequence":"additional","affiliation":[{"name":"National Yang Ming Chiao Tung University, Hsinchu, Taiwan"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-6931-6538","authenticated-orcid":false,"given":"Kai-Chiang","family":"Wu","sequence":"additional","affiliation":[{"name":"National Yang Ming Chiao Tung University, Hsinchu, Taiwan"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2547-4424","authenticated-orcid":false,"given":"Aman","family":"Arora","sequence":"additional","affiliation":[{"name":"Arizona State University, Tempe, AZ, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5734-4221","authenticated-orcid":false,"given":"Diana","family":"Marculescu","sequence":"additional","affiliation":[{"name":"The University of Texas at Austin, Austin, TX, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,2,27]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"2023. ASAP7 PDK and Cell Libraries. https:\/\/github.com\/The-OpenROADProject\/asap7."},{"key":"e_1_3_2_1_2_1","unstructured":"Achronix. 2024. Machine Learning Processor. https:\/\/www.achronix.com\/machine-learning-processor."},{"key":"e_1_3_2_1_3_1","unstructured":"Achronix. 2024. Speedster7t FPGAs: Product Brief. https:\/\/www.achronix.com\/sites\/default\/files\/docs\/Speedster7t_Product_Brief_PB033.pdf."},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1109\/IISWC.2016.7581275"},{"key":"e_1_3_2_1_5_1","unstructured":"AMD. 2023. AI Engine API User Guide. https:\/\/www.xilinx.com\/htmldocs\/xilinx2023_2\/aiengine_api\/aie_api\/doc\/group__group__mmul.html."},{"key":"e_1_3_2_1_6_1","unstructured":"AMD. 2023. AI Engine-ML Programming. https:\/\/github.com\/Xilinx\/Vitis- Tutorials\/blob\/2023.2\/AI_Engine_Development\/AIE-ML\/Design_Tutorials\/01-AIE-ML-programming-and-optimization\/ComputeOptimization.md."},{"key":"e_1_3_2_1_7_1","unstructured":"AMD. 2023. AMD CDNA 3 Architecture. https:\/\/www.amd.com\/content\/ dam\/amd\/en\/documents\/instinct-tech-docs\/white-papers\/amd-cdna-3-whitepaper.pdf."},{"key":"e_1_3_2_1_8_1","unstructured":"AMD. 2024. AI Engine-ML Kernel and Graph Programming Guide (UG1603). https:\/\/docs.amd.com\/r\/en-US\/ug1603-ai-engine-ml-kernel-graph\/Overview?tocId=UXKuGgdu4z0LghCWyYalPw."},{"key":"e_1_3_2_1_9_1","unstructured":"AMD. 2024. AMD Versal\u2122 AI Edge Series VEK280 Evaluation Kit. https:\/\/www.xilinx.com\/products\/boards-and-kits\/vek280.html."},{"key":"e_1_3_2_1_10_1","unstructured":"AMD. 2024. VCK5000 Versal Development Card. https:\/\/www.xilinx.com\/products\/boards-and-kits\/vck5000.html."},{"key":"e_1_3_2_1_11_1","unstructured":"AMD. 2024. Versal Adaptive SoC AIE-ML Architecture Manual (AM020). https: \/\/docs.amd.com\/r\/en-US\/am020-versal-aie-ml\/Overview."},{"key":"e_1_3_2_1_12_1","unstructured":"AMD. 2024. Versal AI Core Series Data Sheet: DC and AC Switching Characteristics (DS957). https:\/\/docs.amd.com\/r\/en-US\/ds957-versal-ai-core\/DSP58-Switching-Characteristics."},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1145\/3529650"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1145\/3431920.3439282"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"crossref","unstructured":"Andrew Boutros Aman Arora and Vaughn Betz. 2024. Field-Programmable Gate Array Architecture for Deep Learning: Survey & Future Directions. arXiv:2404.10076 [cs.AR] https:\/\/arxiv.org\/abs\/2404.10076","DOI":"10.1007\/978-981-97-9314-3_49"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1109\/MCAS.2021.3071607"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICFPT59805.2023.00027"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1145\/3543622.3573044"},{"key":"e_1_3_2_1_19_1","unstructured":"Aiken Cairncross Basile Henry Chris Chalmers Douglas Reid Jonny Shipton Jon Fowler Liz Corrigan and Mike Ashby. 2023. AI Benchmarking on Achronix Speedster\u00ae 7t FPGAs. (2023)."},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1145\/3289602.3293898"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/JETCAS.2019.2910232"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/FPT.2013.6718327"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.mejo.2016.04.006"},{"key":"e_1_3_2_1_24_1","volume-title":"Imagenet: A Large-Scale Hierarchical Image Database. In 2009 IEEE Conference on Computer Vision and Pattern Recognition. Ieee, 248--255","author":"Deng Jia","year":"2009","unstructured":"Jia Deng,Wei Dong, Richard Socher, Li-Jia Li, Kai Li, and Li Fei-Fei. 2009. Imagenet: A Large-Scale Hierarchical Image Database. In 2009 IEEE Conference on Computer Vision and Pattern Recognition. Ieee, 248--255."},{"key":"e_1_3_2_1_25_1","volume-title":"BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding. In North American","author":"Devlin Jacob","year":"2019","unstructured":"Jacob Devlin, Ming-Wei Chang, Kenton Lee, and Kristina Toutanova. 2019. BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding. In North American Chapter of the Association for Computational Linguistics. https:\/\/api.semanticscholar.org\/CorpusID:52967399"},{"key":"e_1_3_2_1_26_1","volume-title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale. ArXiv abs\/2010.11929","author":"Dosovitskiy Alexey","year":"2020","unstructured":"Alexey Dosovitskiy, Lucas Beyer, Alexander Kolesnikov, Dirk Weissenborn, Xiaohua Zhai, Thomas Unterthiner, Mostafa Dehghani, Matthias Minderer, Georg Heigold, Sylvain Gelly, Jakob Uszkoreit, and Neil Houlsby. 2020. An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale. ArXiv abs\/2010.11929 (2020). https:\/\/api.semanticscholar.org\/CorpusID:225039882"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1109\/TVLSI.2022.3197282"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/FCCM.2014.23"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1145\/3352460.3358291"},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1109\/FCCM60383.2024.00016"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2022.3152247"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA.2016.30"},{"key":"e_1_3_2_1_33_1","volume-title":"Proceedings of the 28th International Conference on Neural Information Processing Systems -","volume":"1","author":"Han Song","unstructured":"Song Han, Jeff Pool, John Tran, and William J. Dally. 2015. Learning both weights and connections for efficient neural networks. In Proceedings of the 28th International Conference on Neural Information Processing Systems - Volume 1 (Montreal, Canada) (NIPS'15). MIT Press, Cambridge, MA, USA, 1135--1143."},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1145\/3352460.3358275"},{"key":"e_1_3_2_1_35_1","article-title":"Sparsity in Deep Learning: Pruning and Growth for Efficient Inference and Training in Neural Networks","volume":"22","author":"Hoefler Torsten","year":"2021","unstructured":"Torsten Hoefler, Dan Alistarh, Tal Ben-Nun, Nikoli Dryden, and Alexandra Peste. 2021. Sparsity in Deep Learning: Pruning and Growth for Efficient Inference and Training in Neural Networks. J. Mach. Learn. Res. 22, 1, Article 241 (jan 2021), 124 pages.","journal-title":"J. Mach. Learn. Res."},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW63382.2024.00799"},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1109\/HPEC.2019.8916419"},{"key":"e_1_3_2_1_38_1","unstructured":"Intel. 2020. Intel Agilex Variable Precision DSP Blocks User Guide. https:\/\/www.intel.com\/content\/dam\/altera-www\/global\/en_US\/pdfs\/literature\/ hb\/agilex\/ug-ag-dsp.pdf."},{"key":"e_1_3_2_1_39_1","unstructured":"Intel. 2023. Intel Stratix 10 Embedded Memory User Guide. https:\/\/www.intel.com\/content\/www\/us\/en\/docs\/programmable\/683423\/23--2\/embedded-memory-configurations.html."},{"key":"e_1_3_2_1_40_1","unstructured":"Intel. 2024. Agilex\u2122 5 FPGAs and SoCs Device Data Sheet. https:\/\/www.intel.com\/content\/www\/us\/en\/docs\/programmable\/813918\/current\/agilex-5-fpgasand-socs-device-data-sheet.html."},{"key":"e_1_3_2_1_41_1","unstructured":"Intel. 2024. Agilex\u2122 5 FPGAs: Enhanced DSP with AI Tensor Block. https:\/\/www.intel.com\/content\/www\/us\/en\/content-details\/776602\/agilex-5-fpgas-enhanced-dsp-with-ai-tensor-block.html."},{"key":"e_1_3_2_1_42_1","unstructured":"Intel. 2024. Embedded Memory User Guide: Agilex\u2122 5 FPGAs and SoCs. https:\/\/www.intel.com\/content\/www\/us\/en\/docs\/programmable\/813901\/24--1\/embedded-memory-configurations.html."},{"key":"e_1_3_2_1_43_1","unstructured":"Intel. 2024. Embedded Memory User Guide: Agilex\u2122 7 FPGAs and SoCs. https:\/\/www.intel.com\/content\/www\/us\/en\/docs\/programmable\/683241\/24--2\/embedded-memory-configurations.html."},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.eswa.2023.122666"},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1109\/HPEC49654.2021.9622804"},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.1109\/FPL50879.2020.00031"},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA56546.2023.10071058"},{"key":"e_1_3_2_1_48_1","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA52012.2021.00010"},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.1145\/3140659.3080246"},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"publisher","DOI":"10.1109\/MC.1982.1653825"},{"key":"e_1_3_2_1_51_1","doi-asserted-by":"publisher","DOI":"10.1145\/3431920.3439293"},{"key":"e_1_3_2_1_52_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCAD.2021.3066563"},{"key":"e_1_3_2_1_53_1","doi-asserted-by":"publisher","DOI":"10.1109\/FPL53798.2021.00060"},{"key":"e_1_3_2_1_54_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01167"},{"key":"e_1_3_2_1_55_1","unstructured":"Zhi-Gang Liu Paul N. Whatmough and Matthew Mattina. 2020. Sparse Systolic Tensor Array for Efficient CNN Hardware Acceleration. arXiv:2009.02381 [cs.AR] https:\/\/arxiv.org\/abs\/2009.02381"},{"key":"e_1_3_2_1_56_1","doi-asserted-by":"publisher","DOI":"10.1109\/LCA.2020.2979965"},{"key":"e_1_3_2_1_57_1","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA53966.2022.00049"},{"key":"e_1_3_2_1_58_1","doi-asserted-by":"publisher","DOI":"10.1109\/DAC.2018.8465842"},{"key":"e_1_3_2_1_59_1","doi-asserted-by":"publisher","DOI":"10.1109\/FCCM.2019.00013"},{"key":"e_1_3_2_1_60_1","volume-title":"Jeff Pool, Darko Stosic, Dusan Stosic, Ganesh Venkatesh, Chong Yu, and Paulius Micikevicius.","author":"Mishra Asit","year":"2021","unstructured":"Asit Mishra, Jorge Albericio Latorre, Jeff Pool, Darko Stosic, Dusan Stosic, Ganesh Venkatesh, Chong Yu, and Paulius Micikevicius. 2021. Accelerating Sparse Deep Neural Networks. https:\/\/arxiv.org\/abs\/2104.08378"},{"key":"e_1_3_2_1_61_1","doi-asserted-by":"publisher","DOI":"10.1145\/3388617"},{"key":"e_1_3_2_1_62_1","doi-asserted-by":"publisher","DOI":"10.1109\/MM.2021.3058217"},{"key":"e_1_3_2_1_63_1","unstructured":"NVIDIA. 2020. NVIDIA A100 Tensor Core GPU Architecture. https:\/\/images.nvidia.com\/aem-dam\/en-zz\/Solutions\/data-center\/nvidiaampere-architecture-whitepaper.pdf."},{"key":"e_1_3_2_1_64_1","unstructured":"NVIDIA. 2023. Confidential Compute on NVIDIA Hopper H100. https:\/\/images. nvidia.com\/aem-dam\/en-zz\/Solutions\/data-center\/HCC-Whitepaper-v1.0.pdf."},{"key":"e_1_3_2_1_65_1","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA.2018.00067"},{"key":"e_1_3_2_1_66_1","doi-asserted-by":"publisher","DOI":"10.1109\/FCCM.2019.00015"},{"key":"e_1_3_2_1_67_1","doi-asserted-by":"publisher","DOI":"10.1109\/ISPASS48437.2020.00016"},{"key":"e_1_3_2_1_68_1","doi-asserted-by":"publisher","DOI":"10.1109\/LCA.2019.2924007"},{"key":"e_1_3_2_1_69_1","doi-asserted-by":"crossref","unstructured":"A. Stillmaker and B. Baas. 2017. Scaling equations for the accurate prediction of CMOS device performance from 180 nm to 7 nm. Integration the VLSI Journal 58 (2017) 74--81. http:\/\/vcl.ece.ucdavis.edu\/pubs\/2017.02.VLSIintegration.TechScale\/.","DOI":"10.1016\/j.vlsi.2017.02.002"},{"key":"e_1_3_2_1_70_1","volume-title":"Lin (Eds.)","volume":"33","author":"Sun Xiao","year":"2020","unstructured":"Xiao Sun, Naigang Wang, Chia-Yu Chen, Jiamin Ni, Ankur Agrawal, Xiaodong Cui, Swagath Venkataramani, Kaoutar El Maghraoui, Vijayalakshmi (Viji) Srinivasan, and Kailash Gopalakrishnan. 2020. Ultra-Low Precision 4-bit Training of Deep Neural Networks. In Advances in Neural Information Processing Systems, H. Larochelle, M. Ranzato, R. Hadsell, M.F. Balcan, and H. Lin (Eds.), Vol. 33. Curran Associates, Inc., 1796--1807. https:\/\/proceedings.neurips.cc\/paper_files\/ paper\/2020\/file\/13b919438259814cd5be8cb45877d577-Paper.pdf"},{"key":"e_1_3_2_1_71_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICFPT59805.2023.00016"},{"key":"e_1_3_2_1_72_1","doi-asserted-by":"publisher","DOI":"10.1109\/FCCM60383.2024.00015"},{"key":"e_1_3_2_1_73_1","doi-asserted-by":"publisher","DOI":"10.23919\/DATE58400.2024.10546747"},{"key":"e_1_3_2_1_74_1","volume-title":"Proceedings of the 38th International Conference on Machine Learning (Proceedings of Machine Learning Research","volume":"10357","author":"Touvron Hugo","year":"2021","unstructured":"Hugo Touvron, Matthieu Cord, Matthijs Douze, Francisco Massa, Alexandre Sablayrolles, and Herve Jegou. 2021. Training Data-efficient Image Transformers & Distillation Through Attention. In Proceedings of the 38th International Conference on Machine Learning (Proceedings of Machine Learning Research, Vol. 139), Marina Meila and Tong Zhang (Eds.). PMLR, 10347--10357. https:\/\/proceedings.mlr.press\/v139\/touvron21a.html"},{"key":"e_1_3_2_1_75_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCAD.2022.3172600"},{"key":"e_1_3_2_1_76_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCAD.2017.8203889"},{"key":"e_1_3_2_1_77_1","doi-asserted-by":"publisher","DOI":"10.5555\/3295222.3295349"},{"key":"e_1_3_2_1_78_1","unstructured":"VTR. 2023. Verilog-to-Routing Documentation. https:\/\/docs.verilogtorouting. org\/en\/latest\/vtr\/benchmarks\/."},{"key":"e_1_3_2_1_79_1","doi-asserted-by":"publisher","DOI":"10.1145\/3431388"},{"key":"e_1_3_2_1_80_1","doi-asserted-by":"publisher","DOI":"10.1145\/3061639.3062207"},{"key":"e_1_3_2_1_81_1","doi-asserted-by":"publisher","DOI":"10.1145\/3613424.3623786"},{"key":"e_1_3_2_1_82_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCSI.2021.3074300"},{"key":"e_1_3_2_1_83_1","doi-asserted-by":"publisher","unstructured":"Shuxin Yang Chenchen Ding Mingqiang Huang Kai Li Chenghao Li Zikun Wei Sixiao Huang Jingyao Dong Liuyang Zhang and Hao Yu. 2024. LAMPS: A Layer-wised Mixed-Precision-and-Sparsity Accelerator for NAS-Optimized CNNs on FPGA. In 2024 IEEE 32nd Annual International Symposium on Field- Programmable Custom Computing Machines (FCCM). 90--96. https:\/\/doi.org\/10.1109\/FCCM60383.2024.00019","DOI":"10.1109\/FCCM60383.2024.00019"},{"key":"e_1_3_2_1_84_1","doi-asserted-by":"publisher","DOI":"10.1145\/3301298"},{"key":"e_1_3_2_1_85_1","doi-asserted-by":"publisher","DOI":"10.1145\/3549937"},{"key":"e_1_3_2_1_86_1","doi-asserted-by":"publisher","DOI":"10.1007\/978--3-031--19775--8_12"},{"key":"e_1_3_2_1_87_1","doi-asserted-by":"publisher","DOI":"10.1109\/MICRO.2016.7783723"},{"key":"e_1_3_2_1_88_1","volume-title":"Oh (Eds.)","volume":"35","author":"Zhang Yuxin","year":"2022","unstructured":"Yuxin Zhang, Mingbao Lin, ZhiHang Lin, Yiting Luo, Ke Li, Fei Chao, Yongjian Wu, and Rongrong Ji. 2022. Learning Best Combination for Efficient N:M Sparsity. In Advances in Neural Information Processing Systems, S. Koyejo, S. Mohamed, A. Agarwal, D. Belgrave, K. Cho, and A. Oh (Eds.), Vol. 35. Curran Associates, Inc., 941--953. https:\/\/proceedings.neurips.cc\/paper_files\/paper\/2022\/file\/06589ec9d86876508600a678f9c8f51d-Paper-Conference.pdf"},{"key":"e_1_3_2_1_89_1","volume-title":"SpArch: Efficient Architecture for Sparse Matrix Multiplication. 2020 IEEE International Symposium on High Performance Computer Architecture (HPCA) (2020","author":"Zhang Zhekai","year":"2050","unstructured":"Zhekai Zhang, Hanrui Wang, Song Han, and William J. Dally. 2020. SpArch: Efficient Architecture for Sparse Matrix Multiplication. 2020 IEEE International Symposium on High Performance Computer Architecture (HPCA) (2020), 261--274. https:\/\/api.semanticscholar.org\/CorpusID:211205022"},{"key":"e_1_3_2_1_90_1","unstructured":"Aojun Zhou Yukun Ma Junnan Zhu Jianbo Liu Zhijie Zhang Kun Yuan Wenxiu Sun and Hongsheng Li. 2021. Learning N:M Fine-grained Structured Sparse Neural Networks From Scratch. arXiv:2102.04010 [cs.CV] https:\/\/arxiv.org\/abs\/2102.04010"},{"key":"e_1_3_2_1_91_1","doi-asserted-by":"publisher","DOI":"10.1145\/3352460.3358269"}],"event":{"name":"FPGA '25: The 2025 ACM\/SIGDA International Symposium on Field Programmable Gate Arrays","location":"Monterey CA USA","acronym":"FPGA '25","sponsor":["SIGDA ACM Special Interest Group on Design Automation"]},"container-title":["Proceedings of the 2025 ACM\/SIGDA International Symposium on Field Programmable Gate Arrays"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3706628.3708867","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3706628.3708867","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,22]],"date-time":"2025-08-22T21:53:58Z","timestamp":1755899638000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3706628.3708867"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,2,27]]},"references-count":91,"alternative-id":["10.1145\/3706628.3708867","10.1145\/3706628"],"URL":"https:\/\/doi.org\/10.1145\/3706628.3708867","relation":{},"subject":[],"published":{"date-parts":[[2025,2,27]]},"assertion":[{"value":"2025-02-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}