{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,10,24]],"date-time":"2025-10-24T16:49:46Z","timestamp":1761324586735,"version":"3.41.0"},"publisher-location":"New York, NY, USA","reference-count":26,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,5,7]],"date-time":"2024-05-07T00:00:00Z","timestamp":1715040000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,5,7]]},"DOI":"10.1145\/3649153.3649194","type":"proceedings-article","created":{"date-parts":[[2024,7,2]],"date-time":"2024-07-02T10:21:29Z","timestamp":1719915689000},"page":"71-79","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":1,"title":["Register Blocking: An Analytical Modelling Approach for Affine Loop Kernels"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0007-3620-4637","authenticated-orcid":false,"given":"Theologos","family":"Anthimopoulos","sequence":"first","affiliation":[{"name":"School of Informatics, Aristotle University of Thessaloniki, Thessaloniki, Greece"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0460-6061","authenticated-orcid":false,"given":"Georgios","family":"Keramidas","sequence":"additional","affiliation":[{"name":"School of Informatics, Aristotle University of Thessaloniki, Thessaloniki, Greece"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3340-3792","authenticated-orcid":false,"given":"Vasilios","family":"Kelefouras","sequence":"additional","affiliation":[{"name":"School of Engineering, Computing and Mathematics, University of Plymouth, Plymouth, United Kingdom"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-4624-6089","authenticated-orcid":false,"given":"Iakovos","family":"Stamoulis","sequence":"additional","affiliation":[{"name":"Think Silicon, S.A. An Applied Materials Company, Patras, Greece"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2024,7,2]]},"reference":[{"key":"e_1_3_2_1_1_1","author":"Acharya A.","year":"2020","unstructured":"A. Acharya, U. Bondhugula, and A. Cohen. Effective Loop Fusion in Polyhedral Compilation Using Fusion Conflict Graphs. Transactions on Architecture and Code Optimization, 2020","journal-title":"Transactions on Architecture and Code Optimization"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1145\/1375581.1375595"},{"key":"e_1_3_2_1_3_1","author":"Carr S.","year":"1994","unstructured":"S. Carr and K. Kennedy. Improving the Ratio of Memory Operations to Floating-Point Operations in Loops. Transactions on Programming Languages and Systems, 1994","journal-title":"Transactions on Programming Languages and Systems"},{"key":"e_1_3_2_1_4_1","volume-title":"Unroll-and-Jam Using Uniformly Generated Sets. International Symposium on Microarchitecture","author":"Carr S.","year":"1997","unstructured":"S. Carr and Y. Guan. Unroll-and-Jam Using Uniformly Generated Sets. International Symposium on Microarchitecture, 1997"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1109\/CGO.2005.10"},{"key":"e_1_3_2_1_6_1","volume-title":"International Conference on Parallel Computing","author":"Herruzo E.","year":"2006","unstructured":"E. Herruzo, G. Bandera, E.L. Zapata, and O. Plata. Reducing Cache Misses by Loop Reordering. International Conference on Parallel Computing, 2006"},{"key":"e_1_3_2_1_7_1","author":"Kelefouras V.","year":"2019","unstructured":"V. Kelefouras and K. Djemame. A Methodology Correlating Code Optimizations with Data Memory Accesses, Execution Time, and Energy Consumption. Journal of Supercomputing, 2019","journal-title":"Journal of Supercomputing"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPDS.2023.3322037"},{"key":"e_1_3_2_1_9_1","volume-title":"Model-Driven Transformations for Multi- and Many-Core CPUs. International Conference on Programming Language Design and Implementation","author":"Kong M.","year":"2019","unstructured":"M. Kong and L.N. Pouchet. Model-Driven Transformations for Multi- and Many-Core CPUs. International Conference on Programming Language Design and Implementation, 2019"},{"key":"e_1_3_2_1_10_1","volume-title":"A Performance Vocabulary for Affine Loop Transformations. arXiv preprint arXiv:1811.06043","author":"Kong M.","year":"2018","unstructured":"M. Kong and L.N. Pouchet. A Performance Vocabulary for Affine Loop Transformations. arXiv preprint arXiv:1811.06043, 2018"},{"key":"e_1_3_2_1_11_1","volume-title":"CMSIS-NN: Efficient Neural Network Kernels for Arm Cortex-M CPUs. arXiv preprint arXiv:1801.06601","author":"Lai L.","year":"2018","unstructured":"L. Lai, N. Suda, and V. Chandra. CMSIS-NN: Efficient Neural Network Kernels for Arm Cortex-M CPUs. arXiv preprint arXiv:1801.06601, 2018"},{"key":"e_1_3_2_1_12_1","volume-title":"oneDNN Graph Compiler: A Hybrid Approach for High-Performance Deep Learning Compilation. arXiv preprint arXiv:2301.01333","author":"Li J.","year":"2023","unstructured":"J. Li, Z. Qin, Y. Mei, J. Cui, Y. Song, C. Chen, Y. Zhang, L. Du, X. Cheng, B. Jin, J. Ye, E. Lin, and D. Lavery. oneDNN Graph Compiler: A Hybrid Approach for High-Performance Deep Learning Compilation. arXiv preprint arXiv:2301.01333, 2023"},{"key":"e_1_3_2_1_13_1","volume-title":"Research of Register Pressure Aware Loop Unrolling Optimizations for Compiler. MATEC Web of Conferences","author":"Liu X.","year":"2018","unstructured":"X. Liu, L. Ding, Y. Li, G. Chen, and J. Du. Research of Register Pressure Aware Loop Unrolling Optimizations for Compiler. MATEC Web of Conferences, 2018"},{"key":"e_1_3_2_1_14_1","unstructured":"LLVM Compiler: https:\/\/github.com\/LLVM\/LLVM-project\/issues\/38004"},{"key":"e_1_3_2_1_15_1","volume-title":"Encyclopedia of Parallel Computing","author":"Meister B.","year":"2011","unstructured":"B. Meister, N. Vasilache, D. Wohlford, M. Baskaran, A. Leung, and R. Lethin. R-stream Compiler. In Encyclopedia of Parallel Computing, 2011"},{"key":"e_1_3_2_1_16_1","unstructured":"Orio Tool: https:\/\/github.com\/brnorris03\/Orio"},{"key":"e_1_3_2_1_17_1","unstructured":"Pluto Tool: https:\/\/Pluto-compiler.sourceforge.net\/"},{"key":"e_1_3_2_1_18_1","volume-title":"Pruning and Optimization. International Symposium on Principles of Programming Languages","author":"Pouchet L.N.","year":"2011","unstructured":"L.N. Pouchet, U. Bondhugula, C. Bastoul, A. Cohen, J. Ramanujam, P. Sadayappan, and N. Vasilache. Loop Transformations: Convexity, Pruning and Optimization. International Symposium on Principles of Programming Languages, 2011"},{"key":"e_1_3_2_1_19_1","unstructured":"Polly Tool: https:\/\/polly.LLVM.org\/docs\/UsingPollyWithClang.html"},{"key":"e_1_3_2_1_20_1","unstructured":"Github url Register Blocking Source-to-Source:https:\/\/github.com\/Theoo1997\/RB_s2s"},{"key":"e_1_3_2_1_21_1","volume-title":"Journal of Parallel Programming","author":"Sarkar V.","year":"2001","unstructured":"V. Sarkar. Optimized Unrolling of Nested Loops. Journal of Parallel Programming, 2001"},{"key":"e_1_3_2_1_22_1","unstructured":"Valgrind Tool: https:\/\/valgrind.org\/"},{"key":"e_1_3_2_1_23_1","volume-title":"International Workshop on Polyhedral Compilation Techniques","author":"Vasilache N.","year":"2012","unstructured":"N. Vasilache, B. Meister, M. Baskaran, and R. Lethin. Joint Scheduling and Layout Optimization to Enable Multi-Level Vectorization. International Workshop on Polyhedral Compilation Techniques, 2012"},{"key":"e_1_3_2_1_24_1","volume-title":"Register Tiling for Unstructured Sparsity in Neural Network Inference. International Conference on Programming Languages","author":"Wilkinson L.","year":"2023","unstructured":"L. Wilkinson, K. Cheshmi, and M.M. Dehnavi. Register Tiling for Unstructured Sparsity in Neural Network Inference. International Conference on Programming Languages, 2023"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1145\/3570641"},{"key":"e_1_3_2_1_26_1","volume-title":"Modeling the Conflicting Demands of Parallelism and Temporal\/Spatial Locality in Affine Scheduling. International Conference on Compiler Construction","author":"Zinenko O.","year":"2018","unstructured":"O. Zinenko, S. Verdoolaege, C. Reddy, J. Shirako, T. Grosser, V. Sarkar, and A. Cohen. Modeling the Conflicting Demands of Parallelism and Temporal\/Spatial Locality in Affine Scheduling. International Conference on Compiler Construction, 2018"}],"event":{"name":"CF '24: 21st ACM International Conference on Computing Frontiers","sponsor":["SIGMICRO ACM Special Interest Group on Microarchitectural Research and Processing"],"location":"Ischia Italy","acronym":"CF '24"},"container-title":["Proceedings of the 21st ACM International Conference on Computing Frontiers"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3649153.3649194","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3649153.3649194","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T22:50:02Z","timestamp":1750287002000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3649153.3649194"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,5,7]]},"references-count":26,"alternative-id":["10.1145\/3649153.3649194","10.1145\/3649153"],"URL":"https:\/\/doi.org\/10.1145\/3649153.3649194","relation":{},"subject":[],"published":{"date-parts":[[2024,5,7]]},"assertion":[{"value":"2024-07-02","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}