{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,9]],"date-time":"2026-06-09T08:44:21Z","timestamp":1780994661033,"version":"3.54.1"},"publisher-location":"New York, NY, USA","reference-count":94,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,6,21]]},"DOI":"10.1145\/3695053.3731061","type":"proceedings-article","created":{"date-parts":[[2025,6,20]],"date-time":"2025-06-20T16:46:17Z","timestamp":1750437977000},"page":"1627-1640","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":1,"title":["NUPEA: Optimizing Critical Loads on Spatial Dataflow Architectures via Non-Uniform Processing-Element Access"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-0656-4726","authenticated-orcid":false,"given":"Souradip","family":"Ghosh","sequence":"first","affiliation":[{"name":"Efficient Computer, Pittsburgh, PA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-5408-120X","authenticated-orcid":false,"given":"Graham","family":"Gobieski","sequence":"additional","affiliation":[{"name":"Efficient Computer, Pittsburgh, PA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8902-2518","authenticated-orcid":false,"given":"Keyi","family":"Zhang","sequence":"additional","affiliation":[{"name":"Efficient Computer, San Jose, CA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4130-1099","authenticated-orcid":false,"given":"Brandon","family":"Lucia","sequence":"additional","affiliation":[{"name":"Efficient Computer, Pittsburgh, PA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6301-714X","authenticated-orcid":false,"given":"Nathan","family":"Beckmann","sequence":"additional","affiliation":[{"name":"Efficient Computer, 0000-0001-6301-714X, PA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8483-3824","authenticated-orcid":false,"given":"Tony","family":"Nowatzki","sequence":"additional","affiliation":[{"name":"Efficient Computer, Los Angeles, CA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,6,20]]},"reference":[{"key":"e_1_3_3_1_2_2","doi-asserted-by":"publisher","DOI":"10.1145\/2749469.2750386"},{"key":"e_1_3_3_1_3_2","doi-asserted-by":"publisher","DOI":"10.1145\/2749469.2750385"},{"key":"e_1_3_3_1_4_2","doi-asserted-by":"publisher","DOI":"10.1145\/3579990.3580020"},{"key":"e_1_3_3_1_5_2","unstructured":"Arm. 2025. CMSIS DSP Software Library. https:\/\/arm-software.github.io\/CMSIS-DSP\/latest\/index.html\/."},{"key":"e_1_3_3_1_6_2","doi-asserted-by":"crossref","unstructured":"Hadi Asghari-Moghaddam Amin Farmahini-Farahani Katherine Morrow Jung\u00a0Ho Ahn and Nam\u00a0Sung Kim. 2016. Near-DRAM acceleration with single-ISA heterogeneous processing in standard memory modules. IEEE Micro 36 1 (2016) 24\u201334.","DOI":"10.1109\/MM.2016.8"},{"key":"e_1_3_3_1_7_2","unstructured":"Colby Banbury Vijay\u00a0Janapa Reddi Peter Torelli Jeremy Holleman Nat Jeffries Csaba Kiraly Pietro Montino David Kanter Sebastian Ahmed Danilo Pau Urmish Thakker Antonio Torrini Peter Warden Jay Cordaro Giuseppe\u00a0Di Guglielmo Javier Duarte Stephen Gibellini Videet Parekh Honson Tran Nhan Tran Niu Wenxu and Xu Xuesong. 2021. MLPerf Tiny Benchmark. arxiv:https:\/\/arXiv.org\/abs\/2106.07597\u00a0[cs.LG] https:\/\/arxiv.org\/abs\/2106.07597"},{"key":"e_1_3_3_1_8_2","doi-asserted-by":"publisher","DOI":"10.1145\/3503222.3507772"},{"key":"e_1_3_3_1_9_2","doi-asserted-by":"publisher","DOI":"10.1109\/MICRO56248.2022.00083"},{"key":"e_1_3_3_1_10_2","unstructured":"Scott Beamer Krste Asanovi\u0107 and David Patterson. 2017. The GAP Benchmark Suite. arxiv:https:\/\/arXiv.org\/abs\/1508.03619\u00a0[cs.DC] https:\/\/arxiv.org\/abs\/1508.03619"},{"key":"e_1_3_3_1_11_2","doi-asserted-by":"publisher","DOI":"10.1109\/MICRO.2006.10"},{"key":"e_1_3_3_1_12_2","doi-asserted-by":"publisher","unstructured":"Nathan Beckmann Brandon Lucia Graham Gobieski Tony Nowatzki Thomas Jackson Gu\u00e9nol\u00e9 Lallement Keyi Zhang Amolak Nagi Atharv Sathe and Harsh Desai. 2024. Monza: An Energy-Minimal General-Purpose Dataflow SoC for the Internet of Things. IEEE Micro (2024) 1\u20139. 10.1109\/MM.2024.3426611","DOI":"10.1109\/MM.2024.3426611"},{"key":"e_1_3_3_1_13_2","volume-title":"Proc. of the 22nd intl. conf. on Parallel Architectures and Compilation Techniques","author":"Beckmann Nathan","year":"2013","unstructured":"Nathan Beckmann and Daniel Sanchez. 2013. Jigsaw: Scalable Software-Defined Caches. In Proc. of the 22nd intl. conf. on Parallel Architectures and Compilation Techniques."},{"key":"e_1_3_3_1_14_2","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA.2015.7056061"},{"key":"e_1_3_3_1_15_2","doi-asserted-by":"publisher","DOI":"10.1007\/3-540-63465-7_226"},{"key":"e_1_3_3_1_16_2","doi-asserted-by":"publisher","DOI":"10.1145\/1854273.1854350"},{"key":"e_1_3_3_1_17_2","doi-asserted-by":"publisher","DOI":"10.5555\/1023556"},{"key":"e_1_3_3_1_18_2","first-page":"20","volume-title":"Pegasus: An Efficient Intermediate Representation","author":"Budiu Mihai","year":"2002","unstructured":"Mihai Budiu and Seth\u00a0Copen Goldstein. 2002. Pegasus: An Efficient Intermediate Representation. Technical Report CMU-CS-02-107. Carnegie Mellon University. 20 pages. http:\/\/www.cs.cmu.edu\/\u00a0seth\/papers\/budiu-tr02.pdf"},{"key":"e_1_3_3_1_19_2","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA.2006.17"},{"key":"e_1_3_3_1_20_2","doi-asserted-by":"publisher","DOI":"10.1145\/1094811.1094852"},{"key":"e_1_3_3_1_21_2","doi-asserted-by":"publisher","DOI":"10.1109\/MICRO.2006.31"},{"key":"e_1_3_3_1_22_2","doi-asserted-by":"publisher","unstructured":"Ron Cytron Jeanne Ferrante Barry\u00a0K. Rosen Mark\u00a0N. Wegman and F.\u00a0Kenneth Zadeck. 1991. Efficiently Computing Static Single Assignment Form and the Control Dependence Graph. ACM Trans. Program. Lang. Syst. 13 4 (oct 1991) 451\u2013490. 10.1145\/115372.115320","DOI":"10.1145\/115372.115320"},{"key":"e_1_3_3_1_23_2","doi-asserted-by":"publisher","DOI":"10.1145\/3503222.3507706"},{"key":"e_1_3_3_1_24_2","doi-asserted-by":"publisher","DOI":"10.1145\/3352460.3358276"},{"key":"e_1_3_3_1_25_2","doi-asserted-by":"publisher","DOI":"10.1145\/2370816.2370893"},{"key":"e_1_3_3_1_26_2","volume-title":"ISCA","author":"Dennis Jack\u00a0B","year":"1975","unstructured":"Jack\u00a0B Dennis and David\u00a0P Misunas. 1975. A preliminary architecture for a basic data-flow processor. In ISCA."},{"key":"e_1_3_3_1_27_2","doi-asserted-by":"publisher","DOI":"10.1145\/3307650.3322257"},{"key":"e_1_3_3_1_28_2","doi-asserted-by":"publisher","unstructured":"Fabien Gaud Baptiste Lepers Justin Funston Mohammad Dashti Alexandra Fedorova Vivien Qu\u00e9ma Renaud Lachaize and Mark Roth. 2015. Challenges of memory management on modern NUMA systems. Commun. ACM 58 12 (Nov. 2015) 59\u201366. 10.1145\/2814328","DOI":"10.1145\/2814328"},{"key":"e_1_3_3_1_29_2","volume-title":"ISCA","author":"Gobieski Graham","year":"2021","unstructured":"Graham Gobieski, Ahmet\u00a0Oguz Atli, Kenneth Mai, Brandon Lucia, and Nathan Beckmann. 2021. Snafu: an ultra-low-power, energy-minimal CGRA-generation framework and architecture. In ISCA."},{"key":"e_1_3_3_1_30_2","doi-asserted-by":"publisher","DOI":"10.1109\/MICRO56248.2022.00046"},{"key":"e_1_3_3_1_31_2","doi-asserted-by":"crossref","unstructured":"Seth\u00a0Copen Goldstein Herman Schmit Mihai Budiu Srihari Cadambi Matthew Moe and R\u00a0Reed Taylor. 2000. PipeRench: A reconfigurable architecture and compiler. Computer 33 4 (2000).","DOI":"10.1109\/2.839324"},{"key":"e_1_3_3_1_32_2","doi-asserted-by":"crossref","unstructured":"Venkatraman Govindaraju Chen-Han Ho Tony Nowatzki Jatin Chhugani Nadathur Satish Karthikeyan Sankaralingam and Changkyu Kim. 2012. DySER: Unifying functionality and parallelism specialization for energy-efficient computing. IEEE Micro 32 5 (2012).","DOI":"10.1109\/MM.2012.51"},{"key":"e_1_3_3_1_33_2","doi-asserted-by":"publisher","DOI":"10.1145\/3445814.3446749"},{"key":"e_1_3_3_1_34_2","doi-asserted-by":"publisher","DOI":"10.1145\/1555754.1555779"},{"key":"e_1_3_3_1_35_2","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA.2016.46"},{"key":"e_1_3_3_1_36_2","volume-title":"Computer architecture: a quantitative approach","author":"Hennessy John\u00a0L","year":"2011","unstructured":"John\u00a0L Hennessy and David\u00a0A Patterson. 2011. Computer architecture: a quantitative approach. Elsevier."},{"key":"e_1_3_3_1_37_2","doi-asserted-by":"publisher","DOI":"10.1145\/3582016.3582051"},{"key":"e_1_3_3_1_38_2","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA.2019.00018"},{"key":"e_1_3_3_1_39_2","doi-asserted-by":"publisher","DOI":"10.1145\/3061639.3062262"},{"key":"e_1_3_3_1_40_2","doi-asserted-by":"publisher","DOI":"10.1109\/FCCM62733.2025.00062"},{"key":"e_1_3_3_1_41_2","doi-asserted-by":"publisher","DOI":"10.1145\/605397.605420"},{"key":"e_1_3_3_1_42_2","doi-asserted-by":"publisher","DOI":"10.1145\/3126908.3126965"},{"key":"e_1_3_3_1_43_2","doi-asserted-by":"publisher","unstructured":"Fredrik Kjolstad Shoaib Kamil Stephen Chou David Lugato and Saman Amarasinghe. 2017. The tensor algebra compiler. Proc. ACM Program. Lang. 1 OOPSLA Article 77 (oct 2017) 29\u00a0pages. 10.1145\/3133901","DOI":"10.1145\/3133901"},{"key":"e_1_3_3_1_44_2","doi-asserted-by":"publisher","DOI":"10.1145\/3192366.3192379"},{"key":"e_1_3_3_1_45_2","doi-asserted-by":"publisher","DOI":"10.1109\/CGO51591.2021.9370308"},{"key":"e_1_3_3_1_46_2","doi-asserted-by":"publisher","DOI":"10.1145\/291069.291018"},{"key":"e_1_3_3_1_47_2","doi-asserted-by":"publisher","DOI":"10.1145\/1250662.1250707"},{"key":"e_1_3_3_1_48_2","doi-asserted-by":"publisher","DOI":"10.1145\/3373376.3378497"},{"key":"e_1_3_3_1_49_2","doi-asserted-by":"publisher","DOI":"10.1145\/237090.237190"},{"key":"e_1_3_3_1_50_2","doi-asserted-by":"publisher","DOI":"10.5555\/647926.739234"},{"key":"e_1_3_3_1_51_2","doi-asserted-by":"publisher","DOI":"10.1109\/FPGA.1995.242049"},{"key":"e_1_3_3_1_52_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-540-45234-8_7"},{"key":"e_1_3_3_1_53_2","doi-asserted-by":"publisher","DOI":"10.1109\/FPGA.1996.564808"},{"key":"e_1_3_3_1_54_2","doi-asserted-by":"publisher","DOI":"10.1145\/1168857.1168878"},{"key":"e_1_3_3_1_55_2","doi-asserted-by":"publisher","DOI":"10.1145\/143365.143488"},{"key":"e_1_3_3_1_56_2","doi-asserted-by":"publisher","DOI":"10.1145\/2872362.2872363"},{"key":"e_1_3_3_1_57_2","doi-asserted-by":"publisher","DOI":"10.1145\/3466752.3480048"},{"key":"e_1_3_3_1_58_2","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA56546.2023.10071026"},{"key":"e_1_3_3_1_59_2","unstructured":"Rishiyur\u00a0S Nikhil et\u00a0al. 1990. Executing a program on the MIT tagged-token dataflow architecture. IEEE Transactions on computers (1990)."},{"key":"e_1_3_3_1_60_2","doi-asserted-by":"publisher","DOI":"10.1145\/3243176.3243212"},{"key":"e_1_3_3_1_61_2","doi-asserted-by":"publisher","DOI":"10.1145\/3079856.3080255"},{"key":"e_1_3_3_1_62_2","doi-asserted-by":"publisher","DOI":"10.1145\/2749469.2750380"},{"key":"e_1_3_3_1_63_2","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA57654.2024.00024"},{"key":"e_1_3_3_1_64_2","doi-asserted-by":"publisher","DOI":"10.1145\/325164.325117"},{"key":"e_1_3_3_1_65_2","doi-asserted-by":"publisher","DOI":"10.1145\/2485922.2485935"},{"key":"e_1_3_3_1_66_2","doi-asserted-by":"publisher","DOI":"10.1145\/3307650.3322212"},{"key":"e_1_3_3_1_67_2","doi-asserted-by":"publisher","DOI":"10.1145\/3079856.3080256"},{"key":"e_1_3_3_1_68_2","doi-asserted-by":"publisher","DOI":"10.5555\/1025127.1026007"},{"key":"e_1_3_3_1_69_2","doi-asserted-by":"crossref","unstructured":"Alexander Rucker Matthew Vilim Tian Zhao Yaqi Zhang Raghu Prabhakar and Kunle Olukotun. 2021. Capstan: A Vector RDA for Sparsity. arxiv:https:\/\/arXiv.org\/abs\/2104.12760\u00a0[cs.AR]","DOI":"10.1145\/3466752.3480047"},{"key":"e_1_3_3_1_70_2","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA47549.2020.00045"},{"key":"e_1_3_3_1_71_2","doi-asserted-by":"publisher","DOI":"10.1145\/859618.859667"},{"key":"e_1_3_3_1_72_2","doi-asserted-by":"publisher","DOI":"10.1109\/HCS52781.2021.9567306"},{"key":"e_1_3_3_1_73_2","doi-asserted-by":"publisher","DOI":"10.1109\/MICRO61859.2024.00095"},{"key":"e_1_3_3_1_74_2","volume-title":"Proc. of the 49th annual Intl. Symp. on Computer Architecture (Proc. ISCA-49)","author":"Schwedock B.\u00a0C.","year":"2022","unstructured":"B.\u00a0C. Schwedock, P. Yoovidhya, J. Seibert, and N. Beckmann. 2022. t\u00e4k\u014d: A Polymorphic Cache Hierarchy for General-Purpose Optimization of Data Movement. In Proc. of the 49th annual Intl. Symp. on Computer Architecture (Proc. ISCA-49)."},{"key":"e_1_3_3_1_75_2","doi-asserted-by":"publisher","DOI":"10.1145\/3613424.3614283"},{"key":"e_1_3_3_1_76_2","doi-asserted-by":"publisher","unstructured":"H. Singh Ming-Hau Lee Guangming Lu F.J. Kurdahi N. Bagherzadeh and E.M. Chaves\u00a0Filho. 2000. MorphoSys: an integrated reconfigurable system for data-parallel and computation-intensive applications. IEEE Trans. Comput. 49 5 (2000) 465\u2013481. 10.1109\/12.859540","DOI":"10.1109\/12.859540"},{"key":"e_1_3_3_1_77_2","doi-asserted-by":"publisher","DOI":"10.1109\/MICRO.2003.1253203"},{"key":"e_1_3_3_1_78_2","doi-asserted-by":"publisher","unstructured":"Steven Swanson Andrew Schwerin Martha Mercaldi Andrew Petersen Andrew Putnam Ken Michelson Mark Oskin and Susan\u00a0J. Eggers. 2007. The WaveScalar architecture. ACM Trans. Comput. Syst. (2007). 10.1145\/1233307.1233308","DOI":"10.1145\/1233307.1233308"},{"key":"e_1_3_3_1_79_2","doi-asserted-by":"publisher","DOI":"10.1109\/HOTI.2019.00016"},{"key":"e_1_3_3_1_80_2","doi-asserted-by":"publisher","DOI":"10.5555\/647478.727935"},{"key":"e_1_3_3_1_81_2","volume-title":"StreamIT: A Complier for Streaming Applications","author":"Thies William\u00a0F","year":"2001","unstructured":"William\u00a0F Thies, Michael Karczmarek, Michael Gordon, David Maze, Jeremy Wong, Henry Hoffmann, Matthew Brown, and Saman Amarasinghe. 2001. StreamIT: A Complier for Streaming Applications. Technical Report. MIT."},{"key":"e_1_3_3_1_82_2","doi-asserted-by":"publisher","DOI":"10.1145\/3582016.3582026"},{"key":"e_1_3_3_1_83_2","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA51647.2021.00042"},{"key":"e_1_3_3_1_84_2","doi-asserted-by":"publisher","DOI":"10.1145\/3079856.3080214"},{"key":"e_1_3_3_1_85_2","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA52012.2021.00039"},{"key":"e_1_3_3_1_86_2","doi-asserted-by":"publisher","DOI":"10.1145\/2678373.2665703"},{"key":"e_1_3_3_1_87_2","doi-asserted-by":"publisher","DOI":"10.1109\/2.612254"},{"key":"e_1_3_3_1_88_2","doi-asserted-by":"publisher","DOI":"10.1145\/3582016.3582032"},{"key":"e_1_3_3_1_89_2","doi-asserted-by":"publisher","DOI":"10.1145\/3613424.3623778"},{"key":"e_1_3_3_1_90_2","doi-asserted-by":"publisher","DOI":"10.1145\/3307650.3322229"},{"key":"e_1_3_3_1_91_2","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA53966.2022.00032"},{"key":"e_1_3_3_1_92_2","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA51647.2021.00060"},{"key":"e_1_3_3_1_93_2","doi-asserted-by":"publisher","unstructured":"Jian Weng Sihao Liu Dylan Kupsh and Tony Nowatzki. 2022. Unifying Spatial Accelerator Compilation With Idiomatic and Modular Transformations. IEEE Micro 42 5 (2022) 59\u201369. 10.1109\/MM.2022.3189976","DOI":"10.1109\/MM.2022.3189976"},{"key":"e_1_3_3_1_94_2","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA47549.2020.00063"},{"key":"e_1_3_3_1_95_2","unstructured":"Tomofumi Yuki and Louis-Noel Pouchet. 2016. PolyBench 4.2.1: The polyhedral benchmark suite."}],"event":{"name":"ISCA '25: Proceedings of the 52nd Annual International Symposium on Computer Architecture","location":"Tokyo Japan","acronym":"SIGARCH '25","sponsor":["SIGARCH ACM Special Interest Group on Computer Architecture"]},"container-title":["Proceedings of the 52nd Annual International Symposium on Computer Architecture"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3695053.3731061","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,21]],"date-time":"2025-06-21T11:05:00Z","timestamp":1750503900000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3695053.3731061"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,6,20]]},"references-count":94,"alternative-id":["10.1145\/3695053.3731061","10.1145\/3695053"],"URL":"https:\/\/doi.org\/10.1145\/3695053.3731061","relation":{},"subject":[],"published":{"date-parts":[[2025,6,20]]},"assertion":[{"value":"2025-06-20","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}