{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,9,30]],"date-time":"2025-09-30T10:09:47Z","timestamp":1759226987396,"version":"3.40.3"},"publisher-location":"Cham","reference-count":21,"publisher":"Springer Nature Switzerland","isbn-type":[{"type":"print","value":"9783031697654"},{"type":"electronic","value":"9783031697661"}],"license":[{"start":{"date-parts":[[2024,1,1]],"date-time":"2024-01-01T00:00:00Z","timestamp":1704067200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,1,1]],"date-time":"2024-01-01T00:00:00Z","timestamp":1704067200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2024]]},"DOI":"10.1007\/978-3-031-69766-1_4","type":"book-chapter","created":{"date-parts":[[2024,8,25]],"date-time":"2024-08-25T19:02:05Z","timestamp":1724612525000},"page":"47-61","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":2,"title":["Exploring Processor Micro-architectures Optimised for\u00a0BLAS3 Micro-kernels"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-0035-244X","authenticated-orcid":false,"given":"Stepan","family":"Nassyr","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7296-7817","authenticated-orcid":false,"given":"Dirk","family":"Pleiter","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,8,26]]},"reference":[{"key":"4_CR1","doi-asserted-by":"publisher","unstructured":"Alaejos, G., et\u00a0al.: Micro-kernels for portable and efficient matrix multiplication in deep learning. J. Supercomput. 79(7), 8124\u20138147 (2023). https:\/\/doi.org\/10.1007\/s11227-022-05003-3","DOI":"10.1007\/s11227-022-05003-3"},{"key":"4_CR2","unstructured":"Amid, A., et\u00a0al.: RISC-V \"V\" Vector Extension Version 1.0 (2021). https:\/\/github.com\/riscv\/riscv-v-spec\/releases\/download\/v1.0\/riscv-v-spec-1.0.pdf"},{"key":"4_CR3","doi-asserted-by":"publisher","unstructured":"Binkert, N., et\u00a0al.: The gem5 Simulator. SIGARCH Comput. Archit. News 39(2), 1-7 (2011). https:\/\/doi.org\/10.1145\/2024716.2024718","DOI":"10.1145\/2024716.2024718"},{"key":"4_CR4","doi-asserted-by":"publisher","unstructured":"Brank, B.: Vector length agnostic SIMD parallelism on modern processor architectures with the focus on Arm\u2019s SVE. Ph.D. thesis, Bergische Universit\u00e4t Wuppertal (2023). https:\/\/doi.org\/10.25926\/BUW\/0-43","DOI":"10.25926\/BUW\/0-43"},{"key":"4_CR5","doi-asserted-by":"publisher","unstructured":"Brank, B., Pleiter, D.: CPU Architecture Modelling and Co-design. In: Bhatele, A., Hammond, J., Baboulin, M., Kruse, C. (eds.) High Performance Computing (2023). https:\/\/doi.org\/10.1007\/978-3-031-32041-5_1","DOI":"10.1007\/978-3-031-32041-5_1"},{"key":"4_CR6","unstructured":"Chen, T., et\u00a0al.: TVM: an automated end-to-end optimizing compiler for deep learning. In: 13th USENIX OSDI Symposium, pp. 578\u2013594 (Oct 2018). https:\/\/www.usenix.org\/conference\/osdi18\/presentation\/chen"},{"key":"4_CR7","doi-asserted-by":"publisher","unstructured":"Goto, K., Geijn, R.A.v.d.: Anatomy of High-Performance Matrix Multiplication. ACM Trans. Math. Softw. 34(3) (2008). https:\/\/doi.org\/10.1145\/1356052.1356053","DOI":"10.1145\/1356052.1356053"},{"key":"4_CR8","doi-asserted-by":"publisher","unstructured":"Haris, J., et\u00a0al.: SECDA: Efficient hardware\/software co-design of FPGA-based DNN accelerators for edge inference. In: 2021 IEEE 33rd International Symposium on Computer Architecture and High Performance Computing (SBAC-PAD), pp. 33\u201343 (2021). https:\/\/doi.org\/10.1109\/SBAC-PAD53543.2021.00015","DOI":"10.1109\/SBAC-PAD53543.2021.00015"},{"key":"4_CR9","doi-asserted-by":"publisher","unstructured":"Heinecke, A., et\u00a0al.: LIBXSMM: accelerating small matrix multiplications by runtime code generation. In: SC 2016, pp. 981\u2013991 (2016). https:\/\/doi.org\/10.1109\/SC.2016.83","DOI":"10.1109\/SC.2016.83"},{"key":"4_CR10","doi-asserted-by":"publisher","unstructured":"Ikarashi, Y., et\u00a0al.: Exocompilation for productive programming of hardware accelerators. In: PLDI 2022, pp. 703-718. PLDI 2022. ACM, New York (2022). https:\/\/doi.org\/10.1145\/3519939.3523446","DOI":"10.1145\/3519939.3523446"},{"key":"4_CR11","doi-asserted-by":"publisher","unstructured":"Low, T.M., et\u00a0al.: Analytical modeling is enough for high-performance BLIS. ACM Trans. Math. Softw. 43(2) (8 2016). https:\/\/doi.org\/10.1145\/2925987","DOI":"10.1145\/2925987"},{"key":"4_CR12","doi-asserted-by":"publisher","unstructured":"Lowe-Power, J., et\u00a0al.: The gem5 Simulator: Version 20.0+ (2020). https:\/\/doi.org\/10.48550\/ARXIV.2007.03152","DOI":"10.48550\/ARXIV.2007.03152"},{"key":"4_CR13","doi-asserted-by":"publisher","unstructured":"Merchant, F., et\u00a0al.: Accelerating BLAS on custom architecture through algorithm-architecture co-design (2016). https:\/\/doi.org\/10.48550\/arXiv.1610.06385","DOI":"10.48550\/arXiv.1610.06385"},{"key":"4_CR14","doi-asserted-by":"publisher","unstructured":"Minervini, F., et\u00a0al.: Vitruvius+: An Area-Efficient RISC-V Decoupled Vector Coprocessor for High Performance Computing Applications. ACM Trans. Archit. Code Optim. 20(2) (3 2023). https:\/\/doi.org\/10.1145\/3575861","DOI":"10.1145\/3575861"},{"key":"4_CR15","doi-asserted-by":"publisher","unstructured":"Nassyr, S., Pleiter, D.: Artifact of the paper: Exploring processor micro-architectures optimised for BLAS3 micro-kernels (June 2024). https:\/\/doi.org\/10.5281\/zenodo.11671717","DOI":"10.5281\/zenodo.11671717"},{"key":"4_CR16","unstructured":"Nassyr, S., et\u00a0al.: Programmatically Reaching the Roof: Automated BLIS Kernel Generator for SVE and RVV. In: RISC-V Summit Europe (2023)"},{"issue":"2","key":"4_CR17","doi-asserted-by":"publisher","first-page":"53","DOI":"10.1109\/MM.2020.2972222","volume":"40","author":"A Pellegrini","year":"2020","unstructured":"Pellegrini, A., et al.: The Arm Neoverse N1 platform: building blocks for the next-gen cloud-to-edge infrastructure SoC. IEEE Micro 40(2), 53\u201362 (2020). https:\/\/doi.org\/10.1109\/MM.2020.2972222","journal-title":"IEEE Micro"},{"issue":"2","key":"4_CR18","doi-asserted-by":"publisher","first-page":"26","DOI":"10.1109\/MM.2017.35","volume":"37","author":"N Stephens","year":"2017","unstructured":"Stephens, N., et al.: The ARM scalable vector extension. IEEE Micro 37(2), 26\u201339 (2017). https:\/\/doi.org\/10.1109\/MM.2017.35","journal-title":"IEEE Micro"},{"key":"4_CR19","doi-asserted-by":"publisher","unstructured":"Van\u00a0Zee, F.G., van\u00a0de Geijn, R.A.: BLIS: a framework for rapidly instantiating BLAS Functionality. ACM Trans. Math. Softw. 41(3) (2015). https:\/\/doi.org\/10.1145\/2764454","DOI":"10.1145\/2764454"},{"key":"4_CR20","doi-asserted-by":"publisher","unstructured":"Xianyi, Z., et\u00a0al.: Model-driven Level 3 BLAS Performance Optimization on Loongson 3A Processor. In: IEEE 18th ICPADS Conference. pp. 684\u2013691 (2012). https:\/\/doi.org\/10.1109\/ICPADS.2012.97","DOI":"10.1109\/ICPADS.2012.97"},{"key":"4_CR21","doi-asserted-by":"publisher","unstructured":"Zaourar, L., et\u00a0al.: Multilevel simulation-based co-design of next generation HPC microprocessors. In: 2021 International PMBS Workshop, pp. 18\u201329 (2021). https:\/\/doi.org\/10.1109\/PMBS54543.2021.00008","DOI":"10.1109\/PMBS54543.2021.00008"}],"container-title":["Lecture Notes in Computer Science","Euro-Par 2024: Parallel Processing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-69766-1_4","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,8,25]],"date-time":"2024-08-25T19:09:13Z","timestamp":1724612953000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-69766-1_4"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024]]},"ISBN":["9783031697654","9783031697661"],"references-count":21,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-69766-1_4","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2024]]},"assertion":[{"value":"26 August 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"Euro-Par","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Parallel Processing","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Madrid","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Spain","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"26 August 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"30 August 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"30","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"europar2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/2024.euro-par.org\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}