{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,18]],"date-time":"2026-08-18T01:44:31Z","timestamp":1787017471035,"version":"build-2736575974"},"publisher-location":"New York, NY, USA","reference-count":94,"publisher":"ACM","funder":[{"DOI":"10.13039\/501100003816","name":"Huawei Technologies","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100003816","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,18]]},"DOI":"10.1145\/3725843.3756067","type":"proceedings-article","created":{"date-parts":[[2025,10,17]],"date-time":"2025-10-17T17:19:56Z","timestamp":1760721596000},"page":"277-291","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":4,"title":["RayN: Ray Tracing Acceleration with Near-memory Computing"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0000-1125-2033","authenticated-orcid":false,"given":"Mohammadreza","family":"Saed","sequence":"first","affiliation":[{"name":"University of British Columbia, Vancouver, Canada"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1732-4314","authenticated-orcid":false,"given":"Prashant J.","family":"Nair","sequence":"additional","affiliation":[{"name":"University of British Columbia, Vancouver, Canada"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1161-692X","authenticated-orcid":false,"given":"Tor M.","family":"Aamodt","sequence":"additional","affiliation":[{"name":"University of British Columbia, Vancouver, Canada"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,10,17]]},"reference":[{"key":"e_1_3_3_2_2_2","volume-title":"GDDR7 SDRAM","unstructured":"[n. d.]. GDDR7 SDRAM. Retrieved Apr 11, 2025 from https:\/\/en.wikipedia.org\/wiki\/GDDR7_SDRAM"},{"key":"e_1_3_3_2_3_2","volume-title":"JEDEC Publishes HBM2 Specification","unstructured":"[n. d.]. JEDEC Publishes HBM2 Specification. Retrieved Apr 3, 2025 from https:\/\/www.anandtech.com\/show\/9969\/jedec-publishes-hbm2-specification"},{"key":"e_1_3_3_2_4_2","volume-title":"NVIDIA Nsight Graphics","unstructured":"[n. d.]. NVIDIA Nsight Graphics. Retrieved Apr 3, 2025 from https:\/\/developer.nvidia.com\/nsight-graphics"},{"key":"e_1_3_3_2_5_2","volume-title":"UPMEM","unstructured":"[n. d.]. UPMEM. Retrieved Apr 3, 2025 from https:\/\/www.upmem.com\/"},{"key":"e_1_3_3_2_6_2","volume-title":"Vulkan","unstructured":"[n. d.]. Vulkan. Retrieved Apr 3, 2025 from https:\/\/www.vulkan.org\/"},{"key":"e_1_3_3_2_7_2","volume-title":"High Bandwidth Memory Will Stack on AI Chips Starting Around 2026 With HBM4","year":"2024","unstructured":"2024. High Bandwidth Memory Will Stack on AI Chips Starting Around 2026 With HBM4. Retrieved Apr 3, 2025 from https:\/\/www.nextbigfuture.com\/2024\/02\/ai-chips-will-force-stacking-of-high-bandwidth-memory-starting-around-2026-with-hbm4.html"},{"key":"e_1_3_3_2_8_2","doi-asserted-by":"publisher","DOI":"10.1145\/2749469.2750386"},{"key":"e_1_3_3_2_9_2","doi-asserted-by":"crossref","unstructured":"Junwhan Ahn Sungjoo Yoo Onur Mutlu and Kiyoung Choi. 2015. PIM-enabled instructions: A low-overhead locality-aware processing-in-memory architecture. ACM SIGARCH Computer Architecture News 43 3S (2015) 336\u2013348.","DOI":"10.1145\/2872887.2750385"},{"key":"e_1_3_3_2_10_2","doi-asserted-by":"publisher","DOI":"10.5555\/1921479.1921497"},{"key":"e_1_3_3_2_11_2","doi-asserted-by":"publisher","DOI":"10.1145\/1289816.1289877"},{"key":"e_1_3_3_2_12_2","doi-asserted-by":"crossref","unstructured":"Ernesto\u00a0Rivera Alvarado and Julio\u00a0Zamora Madrigal. 2023. An evaluation of Kd-Trees vs Bounding Volume Hierarchy (BVH) acceleration structures in modern CPU architectures. Tecnolog\u00eda en Marcha 36 2 (2023) 86\u201398.","DOI":"10.18845\/tm.v36i2.6098"},{"key":"e_1_3_3_2_13_2","volume-title":"RDNA 2 Instruction Set Architecture Reference Guide","year":"2020","unstructured":"AMD. 2020. RDNA 2 Instruction Set Architecture Reference Guide. https:\/\/www.amd.com\/content\/dam\/amd\/en\/documents\/radeon-tech-docs\/instruction-set-architectures\/rdna2-shader-instruction-set-architecture.pdf"},{"key":"e_1_3_3_2_14_2","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA59077.2024.00034"},{"key":"e_1_3_3_2_15_2","doi-asserted-by":"publisher","DOI":"10.1109\/MICRO61859.2024.00079"},{"key":"e_1_3_3_2_16_2","doi-asserted-by":"crossref","unstructured":"Carsten Benthin Radoslaw Drabinski Lorenzo Tessari and Addis Dittebrandt. 2022. PLOC++ Parallel Locally-Ordered Clustering for Bounding Volume Hierarchy Construction Revisited. Proc. Int\u2019l Conf. on Computer Graphics and Interactive Techniques (SIGGRAPH)3 1\u201313.","DOI":"10.1145\/3543867"},{"key":"e_1_3_3_2_17_2","doi-asserted-by":"crossref","unstructured":"Carsten Benthin Daniel Meister Joshua Barczak Rohan Mehalwal John Tsakok and Andrew Kensler. 2024. H-PLOC: Hierarchical Parallel Locally-Ordered Clustering for Bounding Volume Hierarchy Construction. Proceedings of the ACM on Computer Graphics and Interactive Techniques 7 3 (2024) 1\u201314.","DOI":"10.1145\/3675377"},{"key":"e_1_3_3_2_18_2","doi-asserted-by":"publisher","DOI":"10.1145\/3231578.3231581"},{"key":"e_1_3_3_2_19_2","doi-asserted-by":"crossref","unstructured":"Maciej Besta Raghavendra Kanakagiri Grzegorz Kwasniewski Rachata Ausavarungnirun Jakub Ber\u00e1nek Konstantinos Kanellopoulos Kacper Janda Zur Vonarburg-Shmaria Lukas Gianinazzi Ioana Stefan Juan\u00a0G\u00f3mez Luna Jakub Golinowski Marcin Copik Lukas Kapp-Schwoerer Salvatore Di\u00a0Girolamo Nils Blach Marek Konieczny Onur Mutlu and Torsten Hoefler. 2021. SISA: Set-Centric Instruction Set Architecture for Graph Mining on Processing-in-Memory Systems. Proc. IEEE\/ACM Symp. on Microarch. (MICRO) 282\u2013297.","DOI":"10.1145\/3466752.3480133"},{"key":"e_1_3_3_2_20_2","doi-asserted-by":"crossref","unstructured":"Ji\u0159\u00ed Bittner Michal Hapala and Vlastimil Havran. 2015. Incremental BVH construction for ray tracing. Computers & Graphics 135\u2013144.","DOI":"10.1016\/j.cag.2014.12.001"},{"key":"e_1_3_3_2_21_2","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA45697.2020.00072"},{"key":"e_1_3_3_2_22_2","doi-asserted-by":"publisher","DOI":"10.1145\/3613424.3614288"},{"key":"e_1_3_3_2_23_2","doi-asserted-by":"crossref","unstructured":"Guohao Dai Tianhao Huang Yuze Chi Jishen Zhao Guangyu Sun Yongpan Liu Yu Wang Yuan Xie and Huazhong Yang. 2018. GraphH: A processing-in-memory architecture for large-scale graph processing. IEEE Trans. on Computer-Aided Design of Integrated Circuits and Systems (TCAD) 640\u2013653.","DOI":"10.1109\/TCAD.2018.2821565"},{"key":"e_1_3_3_2_24_2","doi-asserted-by":"crossref","unstructured":"Guohao Dai Zhenhua Zhu Tianyu Fu Chiyue Wei Bangyan Wang Xiangyu Li Yuan Xie Huazhong Yang and Yu Wang. 2022. DIMMining: Pruning-Efficient and Parallel Graph Mining on near-Memory-Computing. Proc. IEEE\/ACM Int\u2019l Symp. on Computer Architecture (ISCA) 130\u2013145.","DOI":"10.1145\/3470496.3527388"},{"key":"e_1_3_3_2_25_2","doi-asserted-by":"crossref","unstructured":"Yangdong Deng Yufei Ni Zonghui Li Shuai Mu and Wenjun Zhang. 2017. Toward real-time ray tracing: A survey on hardware acceleration and microarchitecture techniques. ACM Computing Surveys (CSUR) 50 4 (2017) 1\u201341.","DOI":"10.1145\/3104067"},{"key":"e_1_3_3_2_26_2","doi-asserted-by":"crossref","unstructured":"Alexandar Devic Siddhartha\u00a0Balakrishna Rai Anand Sivasubramaniam Ameen Akel Sean Eilert and Justin Eno. 2022. To pim or not for emerging general purpose processing in ddr memory systems. Proc. IEEE\/ACM Int\u2019l Symp. on Computer Architecture (ISCA) 231\u2013244.","DOI":"10.1145\/3470496.3527431"},{"key":"e_1_3_3_2_27_2","volume-title":"2nd Workshop on Near-Data Processing (WoNDP)","author":"Eckert Yasuko","year":"2014","unstructured":"Yasuko Eckert, Nuwan Jayasena, and Gabriel\u00a0H Loh. 2014. Thermal feasibility of die-stacked processing in memory. In 2nd Workshop on Near-Data Processing (WoNDP)."},{"key":"e_1_3_3_2_28_2","doi-asserted-by":"crossref","unstructured":"Jo\u00e3o\u00a0Dinis Ferreira Gabriel Falcao Juan G\u00f3mez-Luna Mohammed Alser Lois Orosa Mohammad Sadrosadati Jeremie\u00a0S Kim Geraldo\u00a0F Oliveira Taha Shahroodi Anant Nori et\u00a0al. 2022. pluto: Enabling massively parallel computation in dram via lookup tables. Proc. IEEE\/ACM Symp. on Microarch. (MICRO) 900\u2013919.","DOI":"10.1109\/MICRO56248.2022.00067"},{"key":"e_1_3_3_2_29_2","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA59077.2024.00054"},{"key":"e_1_3_3_2_30_2","doi-asserted-by":"publisher","DOI":"10.1109\/ISPASS61541.2024.00024"},{"key":"e_1_3_3_2_31_2","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA45697.2020.00071"},{"key":"e_1_3_3_2_32_2","doi-asserted-by":"publisher","DOI":"10.1145\/2492045.2492054"},{"key":"e_1_3_3_2_33_2","doi-asserted-by":"publisher","DOI":"10.1109\/MICRO61859.2024.00080"},{"key":"e_1_3_3_2_34_2","doi-asserted-by":"crossref","unstructured":"Milad Hashemi Khubaib Eiman Ebrahimi Onur Mutlu and Yale\u00a0N Patt. 2016. Accelerating dependent cache misses with an enhanced memory controller. ACM SIGARCH Computer Architecture News 44 3 (2016) 444\u2013455.","DOI":"10.1145\/3007787.3001184"},{"key":"e_1_3_3_2_35_2","doi-asserted-by":"crossref","unstructured":"Milad Hashemi Onur Mutlu and Yale\u00a0N Patt. 2016. Continuous runahead: Transparent hardware acceleration for memory intensive workloads. Proc. IEEE\/ACM Symp. on Microarch. (MICRO) 1\u201312.","DOI":"10.1109\/MICRO.2016.7783764"},{"key":"e_1_3_3_2_36_2","doi-asserted-by":"publisher","DOI":"10.1109\/MICRO50266.2020.00040"},{"key":"e_1_3_3_2_37_2","doi-asserted-by":"publisher","DOI":"10.1145\/3620666.3651380"},{"key":"e_1_3_3_2_38_2","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA.2016.27"},{"key":"e_1_3_3_2_39_2","doi-asserted-by":"crossref","unstructured":"Pingyi Huo Anusha Devulapally Hasan\u00a0Al Maruf Minseo Park Krishnakumar Nair Meena Arunachalam Gulsum\u00a0Gudukbay Akbulut Mahmut\u00a0Taylan Kandemir and Vijaykrishnan Narayanan. 2024. PIFS-Rec: Process-In-Fabric-Switch for Large-Scale Recommendation System Inferences. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2409.16633 (2024).","DOI":"10.1109\/MICRO61859.2024.00052"},{"key":"e_1_3_3_2_40_2","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA57654.2024.00029"},{"key":"e_1_3_3_2_41_2","doi-asserted-by":"publisher","DOI":"10.1109\/IMW.2017.7939084"},{"key":"e_1_3_3_2_42_2","doi-asserted-by":"publisher","DOI":"10.1145\/3466752.3480063"},{"key":"e_1_3_3_2_43_2","doi-asserted-by":"publisher","DOI":"10.1145\/2492045.2492055"},{"key":"e_1_3_3_2_44_2","doi-asserted-by":"crossref","unstructured":"Liu Ke Xuan Zhang Jinin So Jong-Geon Lee Shin-Haeng Kang Sukhan Lee Songyi Han YeonGon Cho Jin\u00a0Hyun Kim Yongsuk Kwon et\u00a0al. 2021. Near-memory processing in action: Accelerating personalized recommendation with axdimm. IEEE Micro 42 1 (2021) 116\u2013127.","DOI":"10.1109\/MM.2021.3097700"},{"key":"e_1_3_3_2_45_2","doi-asserted-by":"crossref","unstructured":"Jin\u00a0Hyun Kim Shin-Haeng Kang Sukhan Lee Hyeonsu Kim Yuhwan Ro Seungwon Lee David Wang Jihyun Choi Jinin So YeonGon Cho et\u00a0al. 2022. Aquabolt-XL HBM2-PIM LPDDR5-PIM with in-memory processing and AXDIMM with acceleration buffer. IEEE Micro 42 3 (2022) 20\u201330.","DOI":"10.1109\/MM.2022.3164651"},{"key":"e_1_3_3_2_46_2","first-page":"45","volume-title":"IEEE Computer architecture letters 15","author":"Kim Yoongu","year":"2015","unstructured":"Yoongu Kim, Weikun Yang, and Onur Mutlu. 2015. Ramulator: A fast and extensible DRAM simulator. In IEEE Computer architecture letters 15. 45\u201349."},{"key":"e_1_3_3_2_47_2","doi-asserted-by":"publisher","DOI":"10.1111\/cgf.12458"},{"key":"e_1_3_3_2_48_2","doi-asserted-by":"publisher","DOI":"10.1109\/ISSCC42613.2021.9365862"},{"key":"e_1_3_3_2_49_2","doi-asserted-by":"publisher","DOI":"10.1111\/j.1467-8659.2009.01377.x"},{"key":"e_1_3_3_2_50_2","unstructured":"Dongjae Lee Bongjoon Hyun Taehun Kim and Minsoo Rhu. 2024. PIM-MMU: A Memory Management Unit for Accelerating Data Transfers in Commercial PIM Systems. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2409.06204 (2024)."},{"key":"e_1_3_3_2_51_2","first-page":"1","volume-title":"2022 IEEE International Solid-State Circuits Conference (ISSCC)","volume":"65","author":"Lee Seongju","year":"2022","unstructured":"Seongju Lee, Kyuyoung Kim, Sanghoon Oh, Joonhong Park, Gimoon Hong, Dongyoon Ka, Kyudong Hwang, Jeongje Park, Kyeongpil Kang, Jungyeon Kim, et\u00a0al. 2022. A 1ynm 1.25 V 8Gb, 16Gb\/s\/pin GDDR6-based accelerator-in-memory supporting 1TFLOPS MAC operation and various activation functions for deep-learning applications. In 2022 IEEE International Solid-State Circuits Conference (ISSCC) , Vol.\u00a065. IEEE, 1\u20133."},{"key":"e_1_3_3_2_52_2","doi-asserted-by":"publisher","DOI":"10.1145\/2492045.2492057"},{"key":"e_1_3_3_2_53_2","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA59077.2024.00033"},{"key":"e_1_3_3_2_54_2","doi-asserted-by":"publisher","DOI":"10.1145\/3620665.3640376"},{"key":"e_1_3_3_2_55_2","doi-asserted-by":"publisher","DOI":"10.1145\/3620666.3651352"},{"key":"e_1_3_3_2_56_2","doi-asserted-by":"publisher","DOI":"10.1145\/3406185"},{"key":"e_1_3_3_2_57_2","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA59077.2024.00036"},{"key":"e_1_3_3_2_58_2","doi-asserted-by":"publisher","DOI":"10.1145\/3466752.3480097"},{"key":"e_1_3_3_2_59_2","doi-asserted-by":"publisher","DOI":"10.1109\/IISWC59245.2023.00011"},{"key":"e_1_3_3_2_60_2","doi-asserted-by":"publisher","DOI":"10.1145\/3650200.3656601"},{"key":"e_1_3_3_2_61_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-1-4842-7185-8"},{"key":"e_1_3_3_2_62_2","first-page":"1345","volume-title":"IEEE transactions on visualization and computer graphics","author":"Meister Daniel","year":"2017","unstructured":"Daniel Meister and Ji\u0159\u00ed Bittner. 2017. Parallel locally-ordered clustering for bounding volume hierarchy construction. In IEEE transactions on visualization and computer graphics. 1345\u20131353."},{"key":"e_1_3_3_2_63_2","first-page":"171","volume-title":"Emerging Computing: From Devices to Systems: Looking Beyond Moore and Von Neumann","author":"Mutlu Onur","year":"2022","unstructured":"Onur Mutlu, Saugata Ghose, Juan G\u00f3mez-Luna, and Rachata Ausavarungnirun. 2022. A modern primer on processing in memory. In Emerging Computing: From Devices to Systems: Looking Beyond Moore and Von Neumann. Springer, 171\u2013243."},{"key":"e_1_3_3_2_64_2","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS54959.2023.00100"},{"key":"e_1_3_3_2_65_2","doi-asserted-by":"crossref","unstructured":"Jae-Ho Nah Hyuck-Joo Kwon Dong-Seok Kim Cheol-Ho Jeong Jinhong Park Tack-Don Han Dinesh Manocha and Woo-Chan Park. 2014. RayCore: A ray-tracing hardware architecture for mobile devices. ACM Transactions on Graphics (TOG) 33 5 (2014) 1\u201315.","DOI":"10.1145\/2629634"},{"key":"e_1_3_3_2_66_2","doi-asserted-by":"publisher","DOI":"10.1145\/2024156.2024194"},{"key":"e_1_3_3_2_67_2","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA.2017.54"},{"key":"e_1_3_3_2_68_2","volume-title":"NVIDIA Turing GPU architecture: Graphics reinvented","year":"2018","unstructured":"NVIDIA. 2018. NVIDIA Turing GPU architecture: Graphics reinvented. https:\/\/www. nvidia.com\/content\/dam\/en-zz\/Solutions\/design- visualization\/technologies\/turing-architecture\/NVIDIA- Turing-Architecture-Whitepaper.pdf"},{"key":"e_1_3_3_2_69_2","volume-title":"NVIDIA AMPERE GA102 GPU ARCHITECTURE","year":"2020","unstructured":"NVIDIA. 2020. NVIDIA AMPERE GA102 GPU ARCHITECTURE. https:\/\/www.nvidia.com\/content\/PDF\/nvidia-ampere-ga-102-gpu-architecture-whitepaper-v2.pdf"},{"key":"e_1_3_3_2_70_2","volume-title":"NVIDIA ADA GPU ARCHITECTURE","year":"2022","unstructured":"NVIDIA. 2022. NVIDIA ADA GPU ARCHITECTURE. https:\/\/images.nvidia.com\/aem-dam\/Solutions\/Data-Center\/l4\/nvidia-ada-gpu-architecture-whitepaper-V2.02.pdf"},{"key":"e_1_3_3_2_71_2","doi-asserted-by":"publisher","DOI":"10.1145\/3620665.3640422"},{"key":"e_1_3_3_2_72_2","doi-asserted-by":"crossref","unstructured":"Ashutosh Pattnaik Xulong Tang Adwait Jog Onur Kayiran Asit\u00a0K. Mishra Mahmut\u00a0T. Kandemir Onur Mutlu and Chita\u00a0R. Das. 2016. Scheduling Techniques for GPU Architectures with Processing-In-Memory Capabilities. International Conference on Parallel Architectures and Compilation 31\u201344.","DOI":"10.1145\/2967938.2967940"},{"key":"e_1_3_3_2_73_2","volume-title":"Physically Based Rendering, Third Edition: From Theory To Implementation","author":"Pharr Matt","year":"2018","unstructured":"Matt Pharr and Greg Humphreys. 2018. Physically Based Rendering, Third Edition: From Theory To Implementation. Morgan Kaufmann Publishers Inc."},{"key":"e_1_3_3_2_74_2","doi-asserted-by":"publisher","DOI":"10.1109\/MICRO56248.2022.00027"},{"key":"e_1_3_3_2_75_2","first-page":"27","volume-title":"Proc. ACM SIGGRAPH\/EUROGRAPHICS Conf. on Graphics hardware (HWWS)","author":"Schmittler J","year":"2002","unstructured":"J Schmittler, I Wald, and P Slusallek. 2002. SaarCOR: a hardware architecture for ray tracing. In Proc. ACM SIGGRAPH\/EUROGRAPHICS Conf. on Graphics hardware (HWWS). 27\u201336."},{"key":"e_1_3_3_2_76_2","unstructured":"Brian\u00a0C Schwedock and Nathan Beckmann. [n. d.]. Leviathan: A Unified System for General-Purpose Near-Data Computing. ([n. d.])."},{"key":"e_1_3_3_2_77_2","doi-asserted-by":"publisher","DOI":"10.1145\/3620666.3651324"},{"key":"e_1_3_3_2_78_2","unstructured":"JEDEC STANDARD. 2023. High Bandwidth Memory DRAM (HBM3). JESD238A. (2023)."},{"key":"e_1_3_3_2_79_2","doi-asserted-by":"crossref","unstructured":"Weiyi Sun Zhaoshi Li Shouyi Yin Shaojun Wei and Leibo Liu. 2021. ABC-DIMM: Alleviating the Bottleneck of Communication in DIMM-based Near-Memory Processing with Inter-DIMM Broadcast. Proc. IEEE\/ACM Int\u2019l Symp. on Computer Architecture (ISCA) 237\u2013250.","DOI":"10.1109\/ISCA52012.2021.00027"},{"key":"e_1_3_3_2_80_2","doi-asserted-by":"publisher","DOI":"10.1111\/cgf.14759"},{"key":"e_1_3_3_2_81_2","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA59077.2024.00052"},{"key":"e_1_3_3_2_82_2","doi-asserted-by":"crossref","unstructured":"Timo Viitanen Matias Koskela Pekka J\u00e4\u00e4skel\u00e4inen Aleksi Tervo and Jarmo Takala. 2018. PLOCTree: A fast high-quality hardware BVH builder. Proc. Int\u2019l Conf. on Computer Graphics and Interactive Techniques (SIGGRAPH) 1\u201319.","DOI":"10.1145\/3233309"},{"key":"e_1_3_3_2_83_2","doi-asserted-by":"publisher","DOI":"10.1109\/RT.2007.4342588"},{"key":"e_1_3_3_2_84_2","doi-asserted-by":"publisher","DOI":"10.1145\/2601097.2601199"},{"key":"e_1_3_3_2_85_2","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA59077.2024.00035"},{"key":"e_1_3_3_2_86_2","doi-asserted-by":"publisher","DOI":"10.1145\/3613424.3623778"},{"key":"e_1_3_3_2_87_2","doi-asserted-by":"crossref","unstructured":"Zhengrong Wang Jian Weng Sihao Liu and Tony Nowatzki. 2022. Near-stream computing: General and transparent near-cache acceleration. Proc. IEEE Symp. on High-Perf. Computer Architecture (HPCA) 331\u2013345.","DOI":"10.1109\/HPCA53966.2022.00032"},{"key":"e_1_3_3_2_88_2","doi-asserted-by":"crossref","unstructured":"Sven Woop J\u00f6rg Schmittler and Philipp Slusallek. 2005. RPU: a programmable ray processing unit for realtime ray tracing. ACM Transactions on Graphics (TOG) 24 3 (2005) 434\u2013444.","DOI":"10.1145\/1073204.1073211"},{"key":"e_1_3_3_2_89_2","doi-asserted-by":"publisher","DOI":"10.1145\/3105762.3105773"},{"key":"e_1_3_3_2_90_2","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA57654.2024.00030"},{"key":"e_1_3_3_2_91_2","doi-asserted-by":"crossref","unstructured":"Mingxing Zhang Youwei Zhuo Chao Wang Mingyu Gao Yongwei Wu Kang Chen Christos Kozyrakis and Xuehai Qian. 2018. GraphP: Reducing communication for PIM-based graph processing with efficient data partition. Proc. IEEE Symp. on High-Perf. Computer Architecture (HPCA) 544\u2013557.","DOI":"10.1109\/HPCA.2018.00053"},{"key":"e_1_3_3_2_92_2","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA59077.2024.00053"},{"key":"e_1_3_3_2_93_2","doi-asserted-by":"crossref","unstructured":"Zhe Zhou Cong Li Fan Yang and Guangyu Suny. 2023. DIMM-Link: Enabling Efficient Inter-DIMM Communication for Near-Memory Processing. Proc. IEEE Symp. on High-Perf. Computer Architecture (HPCA) 302\u2013316.","DOI":"10.1109\/HPCA56546.2023.10071005"},{"key":"e_1_3_3_2_94_2","doi-asserted-by":"publisher","DOI":"10.1145\/3503221.3508409"},{"key":"e_1_3_3_2_95_2","doi-asserted-by":"crossref","unstructured":"Youwei Zhuo Chao Wang Mingxing Zhang Rui Wang Dimin Niu Yanzhi Wang and Xuehai Qian. 2019. Graphq: Scalable pim-based graph processing. IEEE\/ACM International Symposium on Microarchitecture 712\u2013725.","DOI":"10.1145\/3352460.3358256"}],"event":{"name":"MICRO 2025: 58th IEEE\/ACM International Symposium on Microarchitecture","location":"Seoul Korea","acronym":"MICRO 2025","sponsor":["SIGMICRO ACM Special Interest Group on Microarchitectural Research and Processing"]},"container-title":["Proceedings of the 58th IEEE\/ACM International Symposium on Microarchitecture"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3725843.3756067","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,1,26]],"date-time":"2026-01-26T21:44:09Z","timestamp":1769463849000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3725843.3756067"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,17]]},"references-count":94,"alternative-id":["10.1145\/3725843.3756067","10.1145\/3725843"],"URL":"https:\/\/doi.org\/10.1145\/3725843.3756067","relation":{},"subject":[],"published":{"date-parts":[[2025,10,17]]},"assertion":[{"value":"2025-10-17","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}