{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,2]],"date-time":"2026-07-02T13:44:34Z","timestamp":1782999874209,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":54,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,7,5]],"date-time":"2026-07-05T00:00:00Z","timestamp":1783209600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"funder":[{"name":"Funda\u00e7\u00e3o para a Ci\u00eancia e a Tecnologia","award":["UID\/50021\/2025"],"award-info":[{"award-number":["UID\/50021\/2025"]}]},{"name":"Funda\u00e7\u00e3o para a Ci\u00eancia e a Tecnologia","award":["UID\/PRR\/50021\/2025"],"award-info":[{"award-number":["UID\/PRR\/50021\/2025"]}]},{"name":"Funda\u00e7\u00e3o para a Ci\u00eancia e a Tecnologia","award":["UID\/50008\/2025"],"award-info":[{"award-number":["UID\/50008\/2025"]}]},{"name":"Funda\u00e7\u00e3o para a Ci\u00eancia e a Tecnologia","award":["2022.06780.PTDC"],"award-info":[{"award-number":["2022.06780.PTDC"]}]},{"name":"Funda\u00e7\u00e3o para a Ci\u00eancia e a Tecnologia","award":["2022.11626.BD"],"award-info":[{"award-number":["2022.11626.BD"]}]},{"name":"Funda\u00e7\u00e3o para a Ci\u00eancia e a Tecnologia","award":["2025.01517.BD"],"award-info":[{"award-number":["2025.01517.BD"]}]},{"name":"European Union","award":["101189551"],"award-info":[{"award-number":["101189551"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,7,6]]},"DOI":"10.1145\/3797905.3807866","type":"proceedings-article","created":{"date-parts":[[2026,7,2]],"date-time":"2026-07-02T11:50:37Z","timestamp":1782993037000},"page":"119-131","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["S2VEC: Compiler-Driven Stream Specialization for Linearized Vectorization"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-1116-8859","authenticated-orcid":false,"given":"Lu\u00eds","family":"Crespo","sequence":"first","affiliation":[{"name":"INESC-ID, Instituto Superior T\u00e9cnico, Universidade de Lisboa, Lisbon, Portugal"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-7613-7943","authenticated-orcid":false,"given":"Ana","family":"Fernandes","sequence":"additional","affiliation":[{"name":"Instituto de Telecomunicacoes, University of Coimbra, Coimbra, Portugal"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9805-6747","authenticated-orcid":false,"given":"Gabriel","family":"Falcao","sequence":"additional","affiliation":[{"name":"Instituto de Telecomunicacoes, University of Coimbra, Coimbra, Portugal"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8083-4432","authenticated-orcid":false,"given":"Pedro","family":"Tom\u00e1s","sequence":"additional","affiliation":[{"name":"INESC-ID, Instituto Superior T\u00e9cnico, Universidade de Lisboa, Lisbon, Portugal"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2491-4977","authenticated-orcid":false,"given":"Nuno","family":"Roma","sequence":"additional","affiliation":[{"name":"INESC-ID, Instituto Superior T\u00e9cnico, Universidade de Lisboa, Lisbon, Portugal"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0628-2259","authenticated-orcid":false,"given":"Nuno","family":"Neves","sequence":"additional","affiliation":[{"name":"INESC-ID, Instituto Superior Tecnico, Universidade de Lisboa, Lisbon, Portugal"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,7,5]]},"reference":[{"key":"e_1_3_3_2_2_2","doi-asserted-by":"crossref","unstructured":"Neil Adit and Adrian Sampson. 2022. Performance left on the table: An evaluation of compiler autovectorization for RISC-V. IEEE Micro 42 5 (2022) 41\u201348.","DOI":"10.1109\/MM.2022.3184867"},{"key":"e_1_3_3_2_3_2","doi-asserted-by":"crossref","unstructured":"Hossein Amiri and Asadollah Shahbahrami. 2020. SIMD programming using Intel vector extensions. J. Parallel and Distrib. Comput. 135 (2020) 83\u2013100.","DOI":"10.1016\/j.jpdc.2019.09.012"},{"key":"e_1_3_3_2_4_2","doi-asserted-by":"crossref","unstructured":"Nathan Binkert Bradford Beckmann Gabriel Black Steven\u00a0K Reinhardt Ali Saidi Arkaprava Basu Joel Hestness Derek\u00a0R Hower Tushar Krishna Somayeh Sardashti et\u00a0al. 2011. The gem5 simulator. ACM SIGARCH computer architecture news 39 2 (2011) 1\u20137.","DOI":"10.1145\/2024716.2024718"},{"key":"e_1_3_3_2_5_2","doi-asserted-by":"publisher","DOI":"10.1145\/3519939.3523701"},{"key":"e_1_3_3_2_6_2","series-title":"(MICRO 36)","first-page":"141","volume-title":"Proceedings of the 36th Annual IEEE\/ACM International Symposium on Microarchitecture","author":"Ciricescu Silviu","year":"2003","unstructured":"Silviu Ciricescu, Ray Essick, Brian Lucas, Phil May, Kent Moat, Jim Norris, Michael Schuette, and Ali Saidi. 2003. The Reconfigurable Streaming Vector Processor (RSVPTM). In Proceedings of the 36th Annual IEEE\/ACM International Symposium on Microarchitecture(MICRO 36). IEEE Computer Society, USA, 141."},{"key":"e_1_3_3_2_7_2","doi-asserted-by":"publisher","unstructured":"Lu\u00eds Crespo Nuno Neves Pedro Tom\u00e1s and Nuno Roma. 2025. A Survey on Stream-Based Architectures: From Accelerators to CPUs. Proc. IEEE 113 8 (2025) 713\u2013751. 10.1109\/JPROC.2025.3642972","DOI":"10.1109\/JPROC.2025.3642972"},{"key":"e_1_3_3_2_8_2","doi-asserted-by":"publisher","DOI":"10.1145\/3352460.3358276"},{"key":"e_1_3_3_2_9_2","doi-asserted-by":"publisher","DOI":"10.1145\/1048935.1050187"},{"key":"e_1_3_3_2_10_2","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA52012.2021.00025"},{"key":"e_1_3_3_2_11_2","doi-asserted-by":"crossref","unstructured":"Alexandre\u00a0E Eichenberger Peng Wu and Kevin O\u2019brien. 2004. Vectorization for SIMD architectures with alignment constraints. Acm sigplan notices 39 6 (2004) 82\u201393.","DOI":"10.1145\/996893.996853"},{"key":"e_1_3_3_2_12_2","doi-asserted-by":"publisher","unstructured":"Ana Fernandes Lu\u00eds Crespo Nuno Neves Pedro Tom\u00e1s Nuno Roma and Gabriel Falcao. 2025. Functional Validation of the RISC-V Unlimited Vector Extension. IEEE Embedded Systems Letters 17 1 (2025) 2\u20135. 10.1109\/LES.2024.3416820","DOI":"10.1109\/LES.2024.3416820"},{"key":"e_1_3_3_2_13_2","doi-asserted-by":"publisher","DOI":"10.1109\/HPCS.2018.00064"},{"key":"e_1_3_3_2_14_2","doi-asserted-by":"crossref","unstructured":"John\u00a0L Hennessy and David\u00a0A Patterson. 2019. A new golden age for computer architecture. Commun. ACM 62 2 (2019) 48\u201360.","DOI":"10.1145\/3282307"},{"key":"e_1_3_3_2_15_2","doi-asserted-by":"publisher","DOI":"10.1109\/ISSCC.2014.6757323"},{"key":"e_1_3_3_2_16_2","doi-asserted-by":"publisher","DOI":"10.1145\/3579990.3580019"},{"key":"e_1_3_3_2_17_2","doi-asserted-by":"publisher","DOI":"10.5555\/2190025.2190061"},{"key":"e_1_3_3_2_18_2","doi-asserted-by":"crossref","unstructured":"Brucek Khailany William\u00a0J Dally Ujval\u00a0J Kapasi Peter Mattson Jinyung Namkoong John\u00a0D Owens Brian Towles Andrew Chang and Scott Rixner. 2001. Imagine: Media processing with streams. IEEE micro 21 2 (2001) 35\u201346.","DOI":"10.1109\/40.918001"},{"key":"e_1_3_3_2_19_2","doi-asserted-by":"publisher","DOI":"10.1145\/2145816.2145824"},{"key":"e_1_3_3_2_20_2","doi-asserted-by":"crossref","unstructured":"Samuel Larsen and Saman Amarasinghe. 2000. Exploiting superword level parallelism with multimedia instruction sets. Acm Sigplan Notices 35 5 (2000) 145\u2013156.","DOI":"10.1145\/358438.349320"},{"key":"e_1_3_3_2_21_2","doi-asserted-by":"publisher","DOI":"10.1109\/CGO51591.2021.9370308"},{"key":"e_1_3_3_2_22_2","doi-asserted-by":"crossref","unstructured":"Charles\u00a0E Leiserson Neil\u00a0C Thompson Joel\u00a0S Emer Bradley\u00a0C Kuszmaul Butler\u00a0W Lampson Daniel Sanchez and Tao\u00a0B Schardl. 2020. There\u2019s plenty of room at the Top: What will drive computer performance after Moore\u2019s law? Science 368 6495 (2020) eaam9744.","DOI":"10.1126\/science.aam9744"},{"key":"e_1_3_3_2_23_2","doi-asserted-by":"publisher","DOI":"10.1145\/3696443.3708952"},{"key":"e_1_3_3_2_24_2","unstructured":"Louis-Noel Pouchet. 2012. PolyBench: The polyhedral benchmark suite. https:\/\/web.cse.ohio-state.edu\/\u00a0pouchet.2\/software\/polybench\/."},{"key":"e_1_3_3_2_25_2","doi-asserted-by":"publisher","DOI":"10.1109\/PACT.2011.68"},{"key":"e_1_3_3_2_26_2","doi-asserted-by":"crossref","unstructured":"Francesco Minervini Oscar Palomar Osman Unsal Enrico Reggiani Josue Quiroga Joan Marimon Carlos Rojas Roger Figueras Abraham Ruiz Alberto Gonzalez Jonnatan Mendoza Ivan Vargas C\u00e9sar Hernandez Joan Cabre Lina Khoirunisya Mustapha Bouhali Julian Pavon Francesc Moll Mauro Olivieri Mario Kovac Mate Kovac Leon Dragic Mateo Valero and Adrian Cristal. 2023. Vitruvius+: An Area-Efficient RISC-V Decoupled Vector Coprocessor for High Performance Computing Applications. ACM Trans. Archit. Code Optim. 20 2 Article 28 (March 2023) 25\u00a0pages.","DOI":"10.1145\/3575861"},{"key":"e_1_3_3_2_27_2","doi-asserted-by":"publisher","DOI":"10.1145\/3696443.3708963"},{"key":"e_1_3_3_2_28_2","doi-asserted-by":"publisher","unstructured":"Nuno Neves Joao\u00a0Mario Domingos Nuno Roma Pedro Tom\u00e1s and Gabriel Falcao. 2022. Compiling for Vector Extensions With Stream-Based Specialization. IEEE Micro 42 5 (Sept. 2022) 49\u201358. 10.1109\/MM.2022.3173405","DOI":"10.1109\/MM.2022.3173405"},{"key":"e_1_3_3_2_29_2","doi-asserted-by":"crossref","unstructured":"Nuno Neves Pedro Tom\u00e1s and Nuno Roma. 2017. Adaptive in-cache streaming for efficient data management. IEEE Transactions on Very Large Scale Integration (VLSI) Systems 25 7 (2017) 2130\u20132143.","DOI":"10.1109\/TVLSI.2017.2671405"},{"key":"e_1_3_3_2_30_2","doi-asserted-by":"crossref","unstructured":"Nuno Neves Pedro Tom\u00e1s and Nuno Roma. 2020. Compiler-assisted data streaming for regular code structures. IEEE Trans. Comput. 70 3 (2020) 483\u2013494.","DOI":"10.1109\/TC.2020.2990302"},{"key":"e_1_3_3_2_31_2","doi-asserted-by":"publisher","DOI":"10.1145\/3079856.3080255"},{"key":"e_1_3_3_2_32_2","doi-asserted-by":"publisher","DOI":"10.1109\/CGO.2006.25"},{"key":"e_1_3_3_2_33_2","doi-asserted-by":"publisher","DOI":"10.1145\/1454115.1454119"},{"key":"e_1_3_3_2_34_2","doi-asserted-by":"publisher","DOI":"10.1109\/SBAC-PAD.2013.17"},{"key":"e_1_3_3_2_35_2","volume-title":"Loop coalescing: A compiler transformation for parallel machines","author":"Polychronopoulos Constantine\u00a0D","year":"1987","unstructured":"Constantine\u00a0D Polychronopoulos. 1987. Loop coalescing: A compiler transformation for parallel machines. Technical Report. Illinois Univ., Urbana (USA)."},{"key":"e_1_3_3_2_36_2","doi-asserted-by":"publisher","DOI":"10.1109\/IISWC.2014.6983050"},{"key":"e_1_3_3_2_37_2","doi-asserted-by":"crossref","unstructured":"Tiago\u00a0B Rodrigues Alexandre Rodrigues Manuel Goul\u00e3o Pedro Tom\u00e1s and Leonel Sousa. 2025. Accelerating NTT with RISC-V Vector Extension for Fully Homomorphic Encryption. IACR Transactions on Cryptographic Hardware and Embedded Systems 2025 4 (2025) 711\u2013736.","DOI":"10.46586\/tches.v2025.i4.711-736"},{"key":"e_1_3_3_2_38_2","doi-asserted-by":"publisher","DOI":"10.23919\/DATE51398.2021.9474230"},{"key":"e_1_3_3_2_39_2","doi-asserted-by":"publisher","unstructured":"Paul Scheffler Florian Zaruba Fabian Schuiki Torsten Hoefler and Luca Benini. 2023. Sparse Stream Semantic Registers: A Lightweight ISA Extension Accelerating General Sparse Linear Algebra. IEEE Trans. Parallel Distrib. Syst. 34 12 (Dec. 2023) 3147\u20133161. 10.1109\/TPDS.2023.3322029","DOI":"10.1109\/TPDS.2023.3322029"},{"key":"e_1_3_3_2_40_2","doi-asserted-by":"crossref","unstructured":"Fabian Schuiki Michael Schaffner Frank\u00a0K G\u00fcrkaynak and Luca Benini. 2018. A scalable near-memory architecture for training deep neural networks on large in-memory datasets. IEEE Trans. Comput. 68 4 (2018) 484\u2013497.","DOI":"10.1109\/TC.2018.2876312"},{"key":"e_1_3_3_2_41_2","doi-asserted-by":"crossref","unstructured":"Fabian Schuiki Florian Zaruba Torsten Hoefler and Luca Benini. 2020. Stream semantic registers: A lightweight RISC-V ISA extension achieving full compute utilization in single-issue cores. IEEE Trans. Comput. 70 2 (2020) 212\u2013227.","DOI":"10.1109\/TC.2020.2987314"},{"key":"e_1_3_3_2_42_2","doi-asserted-by":"crossref","unstructured":"Nigel Stephens Stuart Biles Matthias Boettcher Jacob Eapen Mbou Eyole Giacomo Gabrielli Matt Horsnell Grigorios Magklis Alejandro Martinez Nathanael Premillieu et\u00a0al. 2017. The ARM scalable vector extension. IEEE micro 37 2 (2017) 26\u201339.","DOI":"10.1109\/MM.2017.35"},{"key":"e_1_3_3_2_43_2","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA51647.2021.00061"},{"key":"e_1_3_3_2_44_2","doi-asserted-by":"publisher","DOI":"10.1145\/3696443.3708929"},{"key":"e_1_3_3_2_45_2","doi-asserted-by":"crossref","unstructured":"Jo\u00e3o Vieira Nuno Roma Gabriel Falcao and Pedro Tom\u00e1s. 2024. Ndpmulator: Enabling full-system simulation for near-data accelerators from caches to dram. IEEE Access 12 (2024) 10349\u201310365.","DOI":"10.1109\/ACCESS.2024.3352924"},{"key":"e_1_3_3_2_46_2","doi-asserted-by":"publisher","DOI":"10.1145\/3582016.3582032"},{"key":"e_1_3_3_2_47_2","doi-asserted-by":"publisher","DOI":"10.1145\/3307650.3322229"},{"key":"e_1_3_3_2_48_2","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA53966.2022.00032"},{"key":"e_1_3_3_2_49_2","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA51647.2021.00060"},{"key":"e_1_3_3_2_50_2","unstructured":"Andrew Waterman and Krste Asanovic. 2019. RISC-V \"V\" Vector Extension. https:\/\/github.com\/riscv\/riscv-v-spec."},{"key":"e_1_3_3_2_51_2","doi-asserted-by":"crossref","unstructured":"Jian Weng Sihao Liu Dylan Kupsh and Tony Nowatzki. 2022. Unifying spatial accelerator compilation with idiomatic and modular transformations. IEEE Micro 42 5 (2022) 59\u201369.","DOI":"10.1109\/MM.2022.3189976"},{"key":"e_1_3_3_2_52_2","doi-asserted-by":"crossref","unstructured":"Wm\u00a0A Wulf and Sally\u00a0A McKee. 1995. Hitting the memory wall: Implications of the obvious. ACM SIGARCH computer architecture news 23 1 (1995) 20\u201324.","DOI":"10.1145\/216585.216588"},{"key":"e_1_3_3_2_53_2","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA52012.2021.00087"},{"key":"e_1_3_3_2_54_2","doi-asserted-by":"crossref","unstructured":"Florian Zaruba Fabian Schuiki Torsten Hoefler and Luca Benini. 2020. Snitch: A tiny pseudo dual-issue processor for area and energy efficient execution of floating-point intensive workloads. IEEE Trans. Comput. 70 11 (2020) 1845\u20131860.","DOI":"10.1109\/TC.2020.3027900"},{"key":"e_1_3_3_2_55_2","doi-asserted-by":"publisher","DOI":"10.1145\/3640537.3641566"}],"event":{"name":"ICS '26: 2026 International Conference on Supercomputing","location":"Belfast United Kingdom","acronym":"ICS '26","sponsor":["SIGHPC ACM Special Interest Group on High Performance Computing, Special Interest Group on High Performance Computing","SIGARCH ACM Special Interest Group on Computer Architecture"]},"container-title":["Proceedings of the 40th ACM International Conference on Supercomputing"],"original-title":[],"deposited":{"date-parts":[[2026,7,2]],"date-time":"2026-07-02T12:59:26Z","timestamp":1782997166000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3797905.3807866"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,5]]},"references-count":54,"alternative-id":["10.1145\/3797905.3807866","10.1145\/3797905"],"URL":"https:\/\/doi.org\/10.1145\/3797905.3807866","relation":{},"subject":[],"published":{"date-parts":[[2026,7,5]]},"assertion":[{"value":"2026-07-05","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}