{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,30]],"date-time":"2026-06-30T21:31:35Z","timestamp":1782855095586,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":171,"publisher":"ACM","funder":[{"name":"Sandia National Laboratories","award":["229310"],"award-info":[{"award-number":["229310"]}]},{"name":"Sandia National Laboratories Academic Alliance Program","award":["N&#x5c;&#x2f;A"],"award-info":[{"award-number":["N&#x5c;&#x2f;A"]}]},{"name":"University of Illinois Center for Advanced Semiconductor Chips with Accelerated Performance","award":["N&#x5c;&#x2f;A"],"award-info":[{"award-number":["N&#x5c;&#x2f;A"]}]},{"name":"University of Illinois DREMES HYBRID Center","award":["N&#x5c;&#x2f;A"],"award-info":[{"award-number":["N&#x5c;&#x2f;A"]}]},{"name":"National Science Foundation","award":["2329096"],"award-info":[{"award-number":["2329096"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,3,22]]},"DOI":"10.1145\/3779212.3790151","type":"proceedings-article","created":{"date-parts":[[2026,3,10]],"date-time":"2026-03-10T13:55:26Z","timestamp":1773150926000},"page":"563-582","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":1,"title":["DARTH-PUM: A Hybrid Processing-Using-Memory Architecture"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0008-5018-8482","authenticated-orcid":false,"given":"Ryan","family":"Wong","sequence":"first","affiliation":[{"name":"University of Illinois Urbana-Champaign, Urbana, Illinois, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0450-0067","authenticated-orcid":false,"given":"Ben","family":"Feinberg","sequence":"additional","affiliation":[{"name":"Sandia National Laboratories, Albuquerque, New Mexico, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9138-0613","authenticated-orcid":false,"given":"Saugata","family":"Ghose","sequence":"additional","affiliation":[{"name":"University of Illinois Urbana-Champaign, Urbana, Illinois, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,3,22]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"crossref","unstructured":"S. Aga S. Jeloka A. Subramaniyan S. Narayanasamy D. Blaauw and R. Das. 2017. Compute Caches. In HPCA.","DOI":"10.1109\/HPCA.2017.21"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"crossref","unstructured":"V. Agrawal T. P. Xiao C. H. Bennett B. Feinberg S. Shetty K. Ramkumar H. Medu K. Thekkekara R. Chettuvetty S. Leshner Z. Luzada L. Hinh T. Phan M. J. Marinella and S. Agarwal. 2022. Subthreshold Operation of SONOS Analog Memory to Enable Accurate Low-Power Neural Network Inference. In IEDM.","DOI":"10.2172\/2006312"},{"key":"e_1_3_2_1_3_1","volume-title":"RAELLA: Reforming the Arithmetic for Efficient, Low-Resolution, and Low-Loss Analog PIM: No Retraining Required!. In ISCA.","author":"Andrulis T.","year":"2023","unstructured":"T. Andrulis, J. S. Emer, and V. Sze. 2023. RAELLA: Reforming the Arithmetic for Efficient, Low-Resolution, and Low-Loss Analog PIM: No Retraining Required!. In ISCA."},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"crossref","unstructured":"S. Angizi Z. He and D. Fan. 2018a. PIMA-Logic: A Novel Processing-in-Memory Architecture for Highly Flexible and Energy-Efficient Logic Computation. In DAC.","DOI":"10.1145\/3195970.3196092"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"crossref","unstructured":"S. Angizi Z. He A. S. Rakin and D. Fan. 2018b. CMP-PIM: An Energy-Efficient Comparator-Based Processing-in-Memory Neural Network Accelerator. In DAC.","DOI":"10.1145\/3195970.3196009"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"crossref","unstructured":"S. Angizi J. Sun W. Zhang and D. Fan. 2019. AlignS: A Processing-in-Memory Accelerator for DNA Short Read Alignment Leveraging SOT-MRAM. In DAC.","DOI":"10.1145\/3316781.3317764"},{"key":"e_1_3_2_1_7_1","volume-title":"J. P. Strachan, K. Roy, and D. S. Milojicic.","author":"Ankit A.","year":"2019","unstructured":"A. Ankit, I. E. Hajj, S. R. Chalamalasetti, G. Ndu, M. Foltin, R. S. Williams, P. Faraboschi, W. m. Hwu, J. P. Strachan, K. Roy, and D. S. Milojicic. 2019. PUMA: A Programmable Ultra-Efficient Memristor-Based Accelerator for Machine Learning Inference. In ASPLOS."},{"key":"e_1_3_2_1_8_1","volume-title":"of Illinois Urbana-Champaign","author":"ARCANA Research Group at Univ.","year":"2024","unstructured":"ARCANA Research Group at Univ. of Illinois Urbana-Champaign. 2024. MASTODON -- GitHub Repository. https:\/\/github.com\/ARCANA-Research\/MASTODON\/."},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"crossref","unstructured":"C. H. Bennett T. P. Xiao R. Dellana B. Feinberg S. Agarwal M. J. Marinella V. Agrawal V. Prabhakar K. Ramkumar L. Hinh S. Saha V. Raghavan and R. Chettuvetty. 2020. Device-Aware Inference Operations in SONOS Non-Volatile Memory Arrays. In IRPS.","DOI":"10.1109\/IRPS45951.2020.9129313"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"crossref","unstructured":"A. Bhattacharjee A. Moitra and P. Panda. 2023. HyDe: A Hybrid PCM\/FeFET\/SRAM Device-Search for Optimizing Area and Energy-Efficiencies in Analog IMC Platforms. JETCAS (Oct. 2023).","DOI":"10.1109\/JETCAS.2023.3327748"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"crossref","unstructured":"M. N. Bojnordi and E. Ipek. 2016. Memristive Boltzmann Machine: A Hardware Accelerator for Combinatorial Optimization and Deep Learning. In HPCA.","DOI":"10.1109\/HPCA.2016.7446049"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"crossref","unstructured":"A. Boroumand S. Ghose Y. Kim R. Ausavarungnirun E. Shiu R. Thakur D. Kim A. Kuusela A. Knies P. Ranganathan and O. Mutlu. 2018. Google Workloads for Consumer Devices: Mitigating Data Movement Bottlenecks. In ASPLOS.","DOI":"10.1145\/3173162.3173177"},{"key":"e_1_3_2_1_13_1","volume-title":"The Arithmetic of Polynomials in a Galois Field. Am J. Math. (Jan","author":"Carlitz L.","year":"1932","unstructured":"L. Carlitz. 1932. The Arithmetic of Polynomials in a Galois Field. Am J. Math. (Jan. 1932)."},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"crossref","unstructured":"M. Cassinerio N. Ciocchini and D. Ielmini. 2013. Logic Computation in Phase Change Materials by Threshold and Memory Switching. Advanced Materials (Aug. 2013).","DOI":"10.1002\/adma.201301940"},{"key":"e_1_3_2_1_15_1","volume":"2024","author":"Chen J.","unstructured":"J. Chen, C. Gao, Y. Lu, Y. Zhang, and J. Shu. 2024c. Ares-Flash: Efficient Parallel Integer Arithmetic Operations Using NAND Flash Memory. In MICRO.","journal-title":"J. Shu."},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1145\/3566097.3567860"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"crossref","unstructured":"X.-J. Chen H.-P. Chen and C.-L. Yang. 2024b. PointCIM: A Computing-in-Memory Architecture for Accelerating Deep Point Cloud Analytics. In MICRO.","DOI":"10.1109\/MICRO61859.2024.00097"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"crossref","unstructured":"Y.-C. Chen S. Ando D. Fujiki S. Takamaeda-Yamazaki and K. Yoshioka. 2024a. OSA-HCIM: On-the-Fly Saliency-Aware Hybrid SRAM CIM With Dynamic Precision Configuration. In ASPDAC.","DOI":"10.1109\/ASP-DAC58780.2024.10473966"},{"key":"e_1_3_2_1_19_1","volume-title":"PRIME: A Novel Processing-in-Memory Architecture for Neural Network Computation in ReRAM-Based Main Memory. In ISCA.","author":"Chi P.","year":"2016","unstructured":"P. Chi, S. Li, C. Xu, T. Zhang, J. Zhao, Y. Liu, Y. Wang, and Y. Xie. 2016. PRIME: A Novel Processing-in-Memory Architecture for Neural Network Computation in ReRAM-Based Main Memory. In ISCA."},{"key":"e_1_3_2_1_20_1","volume-title":"CASCADE: Connecting RRAMs to Extend Analog Dataflow in an End-to-End In-Memory Processing Paradigm. In MICRO.","author":"Chou T.","year":"2019","unstructured":"T. Chou, W. Tang, J. Botimer, and Z. Zhang. 2019. CASCADE: Connecting RRAMs to Extend Analog Dataflow in an End-to-End In-Memory Processing Paradigm. In MICRO."},{"key":"e_1_3_2_1_21_1","unstructured":"B. Dally. 2015. Challenges for Future Computing Systems. Keynote talk at HiPEAC."},{"key":"e_1_3_2_1_22_1","volume-title":"BERT: Pre-Training of Deep Bidirectional Transformers for Language Understanding. arXiv:1810.04805 [cs.CL].","author":"Devlin J.","year":"2019","unstructured":"J. Devlin, M.-W. Chang, K. Lee, and K. Toutanova. 2019. BERT: Pre-Training of Deep Bidirectional Transformers for Language Understanding. arXiv:1810.04805 [cs.CL]."},{"key":"e_1_3_2_1_23_1","unstructured":"Domo Inc. 2023. Data Never Sleeps 11.0. https:\/\/www.domo.com\/learn\/infographic\/data-never-sleeps-11."},{"key":"e_1_3_2_1_24_1","volume-title":"Neural Cache: Bit-Serial In-Cache Acceleration of Deep Neural Networks. In ISCA.","author":"Eckert C.","year":"2018","unstructured":"C. Eckert, X. Wang, J. Wang, A. Subramaniyan, R. Iyer, D. M. Sylvester, D. T. Blaauw, and R. Das. 2018. Neural Cache: Bit-Serial In-Cache Acceleration of Deep Neural Networks. In ISCA."},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"crossref","unstructured":"B. Feinberg U. K. R. Vengalam N. Whitehair S. Wang and E. Ipek. 2018a. Enabling Scientific Computing on Memristive Accelerators. In ISCA.","DOI":"10.1109\/ISCA.2018.00039"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"crossref","unstructured":"B. Feinberg S. Wang and E. Ipek. 2018b. Making Memristive Neural Network Accelerators Reliable. In HPCA.","DOI":"10.1109\/HPCA.2018.00015"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"crossref","unstructured":"B. Feinberg R. Wong T. P. Xiao C. H Bennett J. N. Rohan E. G. Boman M. J. Marinella S. Agarwal and E. Ipek. 2021. An Analog Preconditioner for Solving Linear Systems. In HPCA.","DOI":"10.1109\/HPCA51647.2021.00069"},{"key":"e_1_3_2_1_28_1","unstructured":"B. Feinberg T. P. Xiao C. J. Brinker C. H. Bennett M. J. Marinella and S. Agarwal. 2025. CrossSim: Accuracy Simulation of Analog In-Memory Computing. https:\/\/github.com\/sandialabs\/cross-sim\/"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"crossref","unstructured":"D. Fujiki S. Mahlke and R. Das. 2019. Duality Cache for Data Parallel Acceleration. In ISCA.","DOI":"10.1145\/3307650.3322257"},{"key":"e_1_3_2_1_30_1","volume":"202","author":"Gao C.","unstructured":"C. Gao, X. Xin, Y. Lu, Y. Zhang, J. Yang, and J. Shu. 2021. ParaBit: Processing Parallel Bitwise Operations in NAND Flash Memory Based SSDs. In MICRO.","journal-title":"J. Shu."},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"crossref","unstructured":"F. Gao G. Tziantzioulis and D. Wentzlaff. 2019. ComputeDRAM: In-Memory Compute Using Off-the-Shelf DRAMs. In MICRO.","DOI":"10.1145\/3352460.3358260"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"crossref","unstructured":"F. Gao G. Tziantzioulis and D. Wentzlaff. 2022. FracDRAM: Fractional Values in Off-the-Shelf DRAM. In MICRO.","DOI":"10.1109\/MICRO56248.2022.00066"},{"key":"e_1_3_2_1_33_1","volume-title":"Gemini: A Family of Highly Capable Multimodal Models. arXiv:2312.11805 [cs.CL].","author":"Google Gemini Team","year":"2023","unstructured":"Gemini Team at Google. 2023. Gemini: A Family of Highly Capable Multimodal Models. arXiv:2312.11805 [cs.CL]."},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"crossref","unstructured":"Q. Guo X. Guo Y. Bai and E. ?pek. 2011. A Resistive TCAM Accelerator for Data-Intensive Computing. In MICRO.","DOI":"10.1145\/2155620.2155660"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"crossref","unstructured":"Q. Guo X. Guo R. Patel E. Ipek and E. G Friedman. 2013. AC-DIMM: Associative Computing With STT-MRAM. In ISCA.","DOI":"10.1145\/2485922.2485939"},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"crossref","unstructured":"X. Guo F. Merrikh Bayat M. Bavandpour M. Klachko M. R. Mahmoodi M. Prezioso K. K. Likharev and D. B. Strukov. 2017. Fast Energy-Efficient Robust and Reproducible Mixed-Signal Neuromorphic Classifier Based on Embedded NOR Flash Memory Technology. In IEDM.","DOI":"10.1109\/IEDM.2017.8268341"},{"key":"e_1_3_2_1_37_1","volume-title":"FELIX: Fast and Energy-Efficient Logic in Memory. In ICCAD.","author":"Gupta S.","year":"2018","unstructured":"S. Gupta, M. Imani, and T. Rosing. 2018. FELIX: Fast and Energy-Efficient Logic in Memory. In ICCAD."},{"key":"e_1_3_2_1_38_1","volume-title":"SIMDRAM: A Framework for Bit-Serial SIMD Processing Using DRAM. In ASPLOS.","author":"Hajinazar N.","year":"2021","unstructured":"N. Hajinazar, G. F. Oliveira, S. Gregorio, J. D. Ferreira, N. M. Ghiasi, M. Patel, M. Alser, S. Ghose, J. Gomez-Luna, and O. Mutlu. 2021. SIMDRAM: A Framework for Bit-Serial SIMD Processing Using DRAM. In ASPLOS."},{"key":"e_1_3_2_1_39_1","volume":"201","author":"He K.","unstructured":"K. He, X. Zhang, S. Ren, and J. Sun. 2016. Deep Residual Learning for Image Recognition. In CVPR.","journal-title":"J. Sun."},{"key":"e_1_3_2_1_40_1","volume-title":"Newton: A DRAM-Maker's Accelerator-in-Memory (AiM) Architecture for Machine Learning. In MICRO.","author":"He M.","year":"2020","unstructured":"M. He, C. Song, I. Kim, C. Jeong, S. Kim, I. Park, M. Thottethodi, and T. N. Vijaykumar. 2020. Newton: A DRAM-Maker's Accelerator-in-Memory (AiM) Architecture for Machine Learning. In MICRO."},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"crossref","unstructured":"B. Hoffer N. Wainstein C. M. Neumann E. Pop E. Yalon and S. Kvatinsky. 2022. Stateful Logic Using Phase Change Memory. JXCDC (Nov. 2022).","DOI":"10.21203\/rs.3.rs-1061047\/v1"},{"key":"e_1_3_2_1_42_1","volume-title":"SAGE: Saliency-Aware Grouping for Efficient Mapping of LLMs on Analog Compute-in-Memory. In ICCAD.","author":"Hou Y.","year":"2025","unstructured":"Y. Hou, Z. Liu, G. Gagnon, H. Tsai, K. El Maghraoui, G. W. Burr, and L. Liu. 2025a. SAGE: Saliency-Aware Grouping for Efficient Mapping of LLMs on Analog Compute-in-Memory. In ICCAD."},{"key":"e_1_3_2_1_43_1","volume-title":"NORA: Noise-Optimized Rescaling of LLMs on Analog Compute-in-Memory Accelerators. In DATE.","author":"Hou Y.","year":"2025","unstructured":"Y. Hou, H. Tsai, K. El Maghraoui, T. Gokmen, G. W. Burr, and L. Liu. 2025b. NORA: Noise-Optimized Rescaling of LLMs on Analog Compute-in-Memory Accelerators. In DATE."},{"key":"e_1_3_2_1_44_1","volume-title":"ICE: An Intelligent Cognition Engine With 3D NAND-Based In-Memory Computing for Vector Similarity Search Acceleration. In MICRO.","author":"Hu H.-W.","year":"2022","unstructured":"H.-W. Hu, W.-C. Wang, Y.-H. Chang, Y.-C. Lee, B.-R. Lin, H.-M. Wang, Y.-P. Lin, Y.-M. Huang, C.-Y. Lee, T.-H. Su, C.-C. Hsieh, C.-M. Hu, Y.-T. Lai, C.-K. Chen, H.-S. Chen, H.-P. Li, T.-W. Kuo, M.-F. Chang, K.-C. Wang, C.-H. Hung, and C.-Y. Lu. 2022. ICE: An Intelligent Cognition Engine With 3D NAND-Based In-Memory Computing for Vector Similarity Search Acceleration. In MICRO."},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"crossref","unstructured":"M. Hu J. P. Strachan Z. Li E. M. Grafals N. Davila C. Graves S. Lam N. Ge J. J. Yang and R. S. Williams. 2016b. Dot-Product Engine for Neuromorphic Computing: Programming 1T1M Crossbar to Accelerate Matrix-Vector Multiplication. In DAC.","DOI":"10.1145\/2897937.2898010"},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"crossref","unstructured":"M. Hu J. P. Strachan Z. Li and R. S. Williams. 2016a. Dot-Product Engine as Computing Memory to Accelerate Machine Learning Algorithms. In ISQED.","DOI":"10.1109\/ISQED.2016.7479230"},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"crossref","unstructured":"S. Huang A. Ankit P. Silveira R. Antunes S. R. Chalamalasetti I. El Hajj D. E. Kim G. Aguiar P. Bruel S. Serebryakov C. Xu C. Li P. Faraboschi J. P. Strachan D. Chen K. Roy W.-m. Hwu and D. Milojicic. 2021. Mixed Precision Quantization for ReRAM-Based DNN Inference Accelerators. In ASPDAC.","DOI":"10.1145\/3394885.3431554"},{"key":"e_1_3_2_1_48_1","doi-asserted-by":"crossref","unstructured":"B. Hyun T. Kim D. Lee and M. Rhu. 2024. Pathfinding Future PIM Architectures by Demystifying a Commercial PIM Technology. In HPCA.","DOI":"10.1109\/HPCA57654.2024.00029"},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.1145\/3183713.3190661"},{"key":"e_1_3_2_1_50_1","unstructured":"Intel Corp. 2023. Intel\u00ae Core\u2122 i7-13700 Processor. https:\/\/www.intel.com\/content\/www\/us\/en\/products\/sku\/230490\/intel-core-i713700-processor-30m-cache-up-to-5-20-ghz\/specifications.html"},{"key":"e_1_3_2_1_51_1","doi-asserted-by":"crossref","unstructured":"Z. Jahshan and L. Yavits. 2024. MajorK: Majority Based kmer Matching in Commodity DRAM. CAL (Apr. 2024).","DOI":"10.1109\/LCA.2024.3384259"},{"key":"e_1_3_2_1_52_1","doi-asserted-by":"publisher","DOI":"10.1145\/3297858.3304048"},{"key":"e_1_3_2_1_53_1","volume-title":"MINT: Mixed-Precision RRAM-Based IN-Memory Training Architecture. In ISCAS.","author":"Jiang H.","year":"2020","unstructured":"H. Jiang, S. Huang, X. Peng, and S. Yu. 2020. MINT: Mixed-Precision RRAM-Based IN-Memory Training Architecture. In ISCAS."},{"key":"e_1_3_2_1_54_1","doi-asserted-by":"crossref","unstructured":"H. Jin C. Liu H. Liu R. Luo J. Xu F. Mao and X. Liao. 2022. ReHy: A ReRAM-Based Digital\/Analog Hybrid PIM Architecture for Accelerating CNN Training. TPDS (Nov. 2022).","DOI":"10.1109\/TPDS.2021.3138087"},{"key":"e_1_3_2_1_55_1","volume-title":"Accurate Deep Neural Network Inference Using Computational Phase-Change Memory. Nat. Commun.","volume":"11","author":"Joshi V.","year":"2020","unstructured":"V. Joshi, M. Le Gallo, S. Haefeli, I. Boybat, S. R. Nandakumar, C. Piveteau, M. Dazzi, B. Rajendran, A. Sebastian, and E. Eleftheriou. 2020. Accurate Deep Neural Network Inference Using Computational Phase-Change Memory. Nat. Commun., Vol. 11 (May 2020)."},{"key":"e_1_3_2_1_56_1","doi-asserted-by":"crossref","unstructured":"L. Ke U. Gupta B. Y. Cho D. Brooks V. Chandra U. Diril A. Firoozshahian K. Hazelwood B. Jia H.-H. S. Lee M. Li B. Maher D. Mudigere M. Naumov M. Schatz M. Smelyanskiy X. Wang B. Reagen C.-J. Wu M. Hempstead and X. Zhang. 2020. RecNMP: Accelerating Personalized Recommendation With Near-Memory Processing. In ISCA.","DOI":"10.1109\/ISCA45697.2020.00070"},{"key":"e_1_3_2_1_57_1","volume-title":"Near-Memory Processing in Action: Accelerating Personalized Recommendation With AxDIMM","author":"Ke L.","year":"2022","unstructured":"L. Ke, X. Zhang, J. So, J.-G. Lee, S.-H. Kang, S. Lee, S. Han, Y. Cho, J. H. Kim, Y. Kwon, K. Kim, J. Jung, I. Yun, S. J. Park, H. Park, J. Song, J. Cho, K. Sohn, N. S. Kim, and H.-H. S. Lee. 2022. Near-Memory Processing in Action: Accelerating Personalized Recommendation With AxDIMM. IEEE Micro (Jul. 2022)."},{"key":"e_1_3_2_1_58_1","doi-asserted-by":"crossref","unstructured":"G. Kestor R. Gioiosa D. J. Kerbyson and A. Hoisie. 2013. Quantifying the Energy Cost of Data Movement in Scientific Applications. In IISWC.","DOI":"10.1109\/IISWC.2013.6704670"},{"key":"e_1_3_2_1_59_1","doi-asserted-by":"crossref","unstructured":"R. Khaddam-Aljameh M. Stanisavljevic J. Fornt Mas G. Karunaratne M. Braendli F. Liu A. Singh S. M. M\u00fcller U. Egger A. Petropoulos T. Antonakopoulos K. Brew S. Choi I. Ok F. L. Lie N. Saulnier V. Chan I. Ahsan V. Narayanan S. R. Nandakumar M. Le Gallo P. A. Francese A. Sebastian and E. Eleftheriou. 2021. HERMES Core\u2013a 14nm CMOS and PCM-Based In-Memory Compute Core Using an Array of 300ps\/LSB Linearized CCO-Based ADCs and Local Digital Processing. In VLSI Symposia.","DOI":"10.23919\/VLSICircuits52068.2021.9492362"},{"key":"e_1_3_2_1_60_1","doi-asserted-by":"crossref","unstructured":"R. Khaddam-Aljameh M. Stanisavljevic J. F. Mas G. Karunaratne M. Br\u00e4ndli F. Liu A. Singh S. M. M\u00fcller U. Egger A. Petropoulos T. Antonakopoulos K. Brew S. Choi I. Ok F. L. Lie N. Saulnier V. Chan I. Ahsan V. Narayanan S. R. Nandakumar M. Le Gallo P. A. Francese A. Sebastian and E. Eleftheriou. 2022. HERMES-Core\u2014A 1.59-TOPS\/mm2 PCM on 14-nm CMOS In-Memory Compute Core Using 300-ps\/LSB Linearized CCO-Based ADCs. JSSC (Apr. 2022).","DOI":"10.23919\/VLSICircuits52068.2021.9492362"},{"key":"e_1_3_2_1_61_1","doi-asserted-by":"crossref","unstructured":"A. Khadem D. Fujiki H. Chen Y. Gu N. Talati S. Mahlke and R. Das. 2025. Multi-Dimensional Vector ISA Extension for Mobile In-Cache Computing. In HPCA.","DOI":"10.1109\/HPCA61900.2025.00045"},{"key":"e_1_3_2_1_62_1","volume":"202","author":"Kim H.","unstructured":"H. Kim, S. Song, S. Choi, J. Choe, S. Han, J. Park, J. Lee, and J.-J. Kim. 2025. CrossBit: Bitwise Computing in NAND Flash Memory With Inter-Bitline Data Communication. In MICRO.","journal-title":"J. Kim."},{"key":"e_1_3_2_1_63_1","doi-asserted-by":"crossref","unstructured":"J. H. Kim S.-H. Kang S. Lee H. Kim Y. Ro S. Lee D. Wang J. Choi J. So Y. Cho J. Song J. Cho K. Sohn and N. S. Kim. 2022a. Aquabolt-XL HBM2-PIM LPDDR5-PIM With In-Memory Processing and AXDIMM With Acceleration Buffer. IEEE Micro (May 2022).","DOI":"10.1109\/MM.2022.3164651"},{"key":"e_1_3_2_1_64_1","doi-asserted-by":"crossref","unstructured":"J. H. Kim S.-H. Kang S. Lee H. Kim W. Song Y. Ro S. Lee D. Wang H. Shin B. Phuah J. Choi J. So Y. Cho J. Song J. Choi J. Cho K. Sohn Y. Sohn K. Park and N. S. Kim. 2021b. Aquabolt-XL: Samsung HBM2-PIM With In-Memory Processing for ML Accelerators and Beyond. In HCS.","DOI":"10.1109\/HCS52781.2021.9567191"},{"key":"e_1_3_2_1_65_1","unstructured":"S. Kim A. Gholami Z. Yao M. W. Mahoney and K. Keutzer. 2021a. I-BERT: Integer-only BERT Quantization. In PMLR."},{"key":"e_1_3_2_1_66_1","volume":"202","author":"Kim S.","unstructured":"S. Kim, S. Kim, S. Um, S. Kim, K. Kim, and H.-J. Yoo. 2023. Neuro-CIM: ADC-Less Neuromorphic Computing-in-Memory Processor With Operation Gating\/Stopping and Digital\u2013Analog Networks. JSSC (May 2023).","journal-title":"J. Yoo."},{"key":"e_1_3_2_1_67_1","volume":"2022","author":"Kim Y.","unstructured":"Y. Kim, H. Kim, and J.-J. Kim. 2022b. Extreme Partial-Sum Quantization for Analog Computing-In-Memory Neural Network Accelerators. JETC (Oct. 2022).","journal-title":"J. Kim."},{"key":"e_1_3_2_1_68_1","doi-asserted-by":"crossref","unstructured":"G. Krishnan Z. Wang I. Yeo L. Yang J. Meng M. Liehr R. V. Joshi N. C. Cady D. Fan J.-S. Seo and Y. Cao. 2022. Hybrid RRAM\/SRAM in-Memory Computing for Robust DNN Acceleration. IEEE TCAD (Aug. 2022).","DOI":"10.1109\/ICSICT55466.2022.9963165"},{"key":"e_1_3_2_1_69_1","unstructured":"A. Krizhevsky. 2009. Learning Multiple Layers of Features From Tiny Images. Technical Report. Univ. of Toronto."},{"key":"e_1_3_2_1_70_1","doi-asserted-by":"crossref","unstructured":"L. Kull T. Toifl M. Schmatz P. A. Francese C. Menolfi M. Br\u00e4ndli M. Kossel T. Morf T. M. Andersen and Y. Leblebici. 2013. A 3.1 mW 8b 1.2 GS\/s Single-Channel Asynchronous SAR ADC With Alternate Comparators for Enhanced Speed in 32 nm Digital SOI CMOS. JSSC (Sep. 2013).","DOI":"10.1109\/ISSCC.2013.6487818"},{"key":"e_1_3_2_1_71_1","volume-title":"MAGIC: Memristor-Aided Logic. TCAS II (Sep.","author":"Kvatinsky S.","year":"2014","unstructured":"S. Kvatinsky, D. Belousov, S. Liman, G. Satat, N. Wald, E. G. Friedman, A. Kolodny, and U. C. Weiser. 2014a. MAGIC: Memristor-Aided Logic. TCAS II (Sep. 2014)."},{"key":"e_1_3_2_1_72_1","doi-asserted-by":"crossref","unstructured":"S. Kvatinsky A. Kolodny U. C. Weiser and E. G. Friedman. 2011. Memristor-Based IMPLY Logic design Procedure. In ICCD.","DOI":"10.1109\/ICCD.2011.6081389"},{"key":"e_1_3_2_1_73_1","doi-asserted-by":"crossref","unstructured":"S. Kvatinsky G. Satat N. Wald E. G. Friedman A. Kolodny and U. C. Weiser. 2014b. Memristor-Based Material Implication (IMPLY) Logic: Design Principles and Methodologies. TVLSI (2014).","DOI":"10.1109\/TVLSI.2013.2282132"},{"key":"e_1_3_2_1_74_1","doi-asserted-by":"crossref","unstructured":"C. Lammie Y. Wang F. Ponzina J. Klein H. Benmeziane M. Zapater I. Boybat A. Sebastian G. Ansaloni and D. Atienza. 2025. LionHeart: A Layer-Based Mapping Framework for Heterogeneous Systems With Analog In-Memory Computing Tiles. IEEE Trans. Emerg. Top. Comput. (Mar. 2025).","DOI":"10.1109\/TETC.2025.3546128"},{"key":"e_1_3_2_1_75_1","doi-asserted-by":"crossref","unstructured":"D. Lee B. Hyun T. Kim and M. Rhu. 2024. Analysis of Data Transfer Bottlenecks in Commercial PIM Systems: A Study With UPMEM-PIM. CAL (Apr. 2024).","DOI":"10.1109\/LCA.2024.3387472"},{"key":"e_1_3_2_1_76_1","volume-title":"Seog Chung Seo, and Seong Oun Hwang.","author":"Lee Wai-Kong","year":"2022","unstructured":"Wai-Kong Lee, Hwa Jeong Seo, Seog Chung Seo, and Seong Oun Hwang. 2022. AES-GPU. https:\/\/github.com\/benlwk\/AES-GPU\/"},{"key":"e_1_3_2_1_77_1","volume-title":"Interface: Power, Area and Accuracy Co-Optimization for RRAM Crossbar-Based Mixed-Signal Computing System. In DAC.","author":"Li B.","year":"2015","unstructured":"B. Li, L. Xia, P. Gu, Y. Wang, and H. Yang. 2015. Merging the Interface: Power, Area and Accuracy Co-Optimization for RRAM Crossbar-Based Mixed-Signal Computing System. In DAC."},{"key":"e_1_3_2_1_78_1","volume-title":"Analog Content-Addressable Memories With Memristors. Nat. Commun.","volume":"11","author":"Li C.","year":"2020","unstructured":"C. Li, C. E. Graves, X. Sheng, D. Miller, M. Foltin, G. Pedretti, and J. P. Strachan. 2020a. Analog Content-Addressable Memories With Memristors. Nat. Commun., Vol. 11 (Apr. 2020)."},{"key":"e_1_3_2_1_79_1","doi-asserted-by":"crossref","unstructured":"H. Li H. Jin L. Zheng and X. Liao. 2020b. ReSQM: Accelerating Database Operations Using ReRAM-Based Content Addressable Memory. IEEE TCAD (Nov. 2020).","DOI":"10.1109\/TCAD.2020.3012860"},{"key":"e_1_3_2_1_80_1","volume-title":"ASADI: Accelerating Sparse Attention Using Diagonal-Based In-Situ Computing. In HPCA.","author":"Li H.","year":"2024","unstructured":"H. Li, Z. Li, Z. Bai, and T. Mitra. 2024. ASADI: Accelerating Sparse Attention Using Diagonal-Based In-Situ Computing. In HPCA."},{"key":"e_1_3_2_1_81_1","doi-asserted-by":"publisher","DOI":"10.1145\/3123939.3123977"},{"key":"e_1_3_2_1_82_1","doi-asserted-by":"publisher","DOI":"10.1145\/2897937.2898064"},{"key":"e_1_3_2_1_83_1","doi-asserted-by":"crossref","unstructured":"Y. Li J. Chen L. Wang W. Zhang Z. Guo J. Wang Y. Han Z. Li F. Wang C. Dou X. Xu J. Yang Z. Wang and D. Shang. 2023. An ADC-Less RRAM-Based Computing-in-Memory Macro With Binary CNN for Efficient Edge AI. TCAS-II (Jan. 2023).","DOI":"10.1109\/TCSII.2022.3233396"},{"key":"e_1_3_2_1_84_1","doi-asserted-by":"crossref","unstructured":"C. Liu H. Liu H. Jin X. Liao Y. Zhang Z. Duan J. Xu and H. Li. 2022. ReGNN: A ReRAM-Based Heterogeneous Architecture for General Graph Neural Networks. In DAC.","DOI":"10.1145\/3489517.3530479"},{"key":"e_1_3_2_1_85_1","volume":"202","author":"Liu C.","unstructured":"C. Liu, K. Wu, H. Liu, H. Jin, X. Liao, Z. Duan, J. Xu, H. Li, Y. Zhang, and J. Yang. 2025. A ReRAM-Based Processing-In-Memory Architecture for Hyperdimensional Computing. IEEE TCAD (Feb. 2025).","journal-title":"J. Yang."},{"key":"e_1_3_2_1_86_1","doi-asserted-by":"crossref","unstructured":"M. Liu L. Xia Y. Wang and K. Chakrabarty. 2018. Fault Tolerance for RRAM-Based Matrix Operations. In ITC.","DOI":"10.1109\/TEST.2018.8624687"},{"key":"e_1_3_2_1_87_1","volume-title":"Simulator: Version 20.0. https:\/\/arxiv.org\/abs\/2007.03152.","author":"Lowe-Power J.","year":"2020","unstructured":"J. Lowe-Power, A. M. Ahmad, A. Akram, M. Alian, R. Amslinger, M. Andreozzi, A. Armejach, N. Asmussen, S. Bharadwaj, G. Black, G. Bloom, B. R. Bruce, D. R. Carvalho, J. Castrill\u00f3n, L. Chen, N. Derumigny, S. Diestelhorst, W. Elsasser, M. Fariborz, A. F. Farahani, P. Fotouhi, R. Gambord, J. Gandhi, D. Gope, T. Grass, B. Hanindhito, A. Hansson, S. Haria, A. Harris, T. Hayes, A. Herrera, M. Horsnell, S. A. R. Jafri, R. Jagtap, H. Jang, R. Jeyapaul, T. M. Jones, M. Jung, S. Kannoth, H. Khaleghzadeh, Y. Kodama, T. Krishna, T. Marinelli, C. Menard, A. Mondelli, T. M\u00fcck, O. Naji, K. Nathella, H. Nguyen, N. Nikoleris, L. E. Olson, M. S. Orr, B. Pham, P. Prieto, T. Reddy, A. Roelke, M. Samani, A. Sandberg, J. Setoain, B. Shingarov, M. D. Sinclair, T. Ta, R. Thakur, G. Travaglini, M. Upton, N. Vaish, I. Vougioukas, Z. Wang, N. Wehn, C. Weis, D. A. Wood, H. Yoon, and E. F. Zulian. 2020. The gem5 Simulator: Version 20.0. https:\/\/arxiv.org\/abs\/2007.03152."},{"key":"e_1_3_2_1_88_1","volume-title":"JETCAS","volume":"8","author":"Marinella M. J.","year":"2018","unstructured":"M. J. Marinella, S. Agarwal, A. Hsia, I. Richter, R. Jacobs-Gedrim, J. Niroula, S. J. Plimpton, E. Ipek, and C. D. James. 2018. Multiscale Co-Design Analysis of Energy, Latency, Area, and Accuracy of a ReRAM Analog Neural Training Accelerator. JETCAS, Vol. 8 (Jan. 2018)."},{"key":"e_1_3_2_1_89_1","doi-asserted-by":"crossref","unstructured":"V. Milo F. Anzalone C. Zambelli E. P\u00e9rez M. K. Mahadevaiah \u00d3. G. Ossorio P. Olivo C. Wenger and D. Ielmini. 2021. Optimized Programming Algorithms for Multilevel RRAM in Hardware Neural Networks. In IRPS.","DOI":"10.1109\/IRPS46558.2021.9405119"},{"key":"e_1_3_2_1_90_1","unstructured":"Mythic Inc. [n.d.]. M1076 Analog Matrix Processor. https:\/\/mythic.ai\/products\/m1076-analog-matrix-processor\/."},{"key":"e_1_3_2_1_91_1","volume-title":"Newton: Gravitating Towards the Physical Limits of Crossbar Acceleration","author":"Nag A.","year":"2018","unstructured":"A. Nag, R. Balasubramonian, V. Srikumar, R. Walker, A. Shafiee, J. P. Strachan, and N. Muralimanohar. 2018. Newton: Gravitating Towards the Physical Limits of Crossbar Acceleration. IEEE Micro (Sep. 2018)."},{"key":"e_1_3_2_1_92_1","unstructured":"R. Nair S. F. Antao C. Bertolli P. Bose J. R. Brunheroto T. Chen C.-Y. Cher C. H. A. Costa J. Doi C. Evangelinos B. M. Fleischer T. W. Fox D. S. Gallo L. Grinberg J. A. Gunnels A. C. Jacob P. Jacob H. M. Jacobson T. Karkhanis C. Kim J. H. Moreno J. K. O'Brien M. Ohmacht Y. Park D. A. Prener B. S. Rosenburg K. D. Ryu O. Sallenave M. J. Serrano P. D. M. Siegl K. Sugavanam and Z. Sura. 2015. Active Memory Cube: A Processing-in-Memory Architecture for Exascale Systems. IBM JRD (Mar.-May 2015)."},{"key":"e_1_3_2_1_93_1","unstructured":"National Institute of Standards and Technology. 2001. Advanced Encryption Standard (AES). https:\/\/nvlpubs.nist.gov\/nistpubs\/fips\/nist.fips.197.pdf"},{"key":"e_1_3_2_1_94_1","doi-asserted-by":"crossref","unstructured":"J. Nechvatal E. Barker L. Bassham W. Burr M. Dworkin J. Foti and E. Roback. 2001. Report on the Development of the Advanced Encryption Standard (AES). NIST (Nov. 2001).","DOI":"10.6028\/jres.106.023"},{"key":"e_1_3_2_1_95_1","doi-asserted-by":"crossref","unstructured":"S. Negi U. Saxena D. Sharma and K. Roy. 2025. HCiM: ADC-Less Hybrid Analog-Digital Compute in Memory Accelerator for Deep Learning Workloads. In ASPDAC.","DOI":"10.1145\/3658617.3697572"},{"key":"e_1_3_2_1_96_1","doi-asserted-by":"crossref","unstructured":"S. Negi U. Saxena D. Sharma J. Victor I. Ahmed S. K. Gupta and K. Roy. 2024. Algorithm Hardware Co-Design for ADC-Less Compute In-Memory Accelerator. TCASAI (Dec 2024).","DOI":"10.1109\/TCASAI.2024.3495587"},{"key":"e_1_3_2_1_97_1","unstructured":"NVIDIA Corp. [n.d.]. GeForce RTX 4090. https:\/\/www.nvidia.com\/en-us\/geforce\/graphics-cards\/40-series\/rtx-4090\/"},{"key":"e_1_3_2_1_98_1","doi-asserted-by":"crossref","unstructured":"A. Olgun J. G\u00f3mez-Luna K. Kanellopoulos B. Salami H. Hassan O. Ergin and O. Mutlu. 2022. PiDRAM: A Holistic End-to-End FPGA-Based Framework for Processing-in-DRAM. TACO (Nov. 2022).","DOI":"10.1145\/3563697"},{"key":"e_1_3_2_1_99_1","doi-asserted-by":"crossref","unstructured":"A. Olgun H. Hassan A. G. Ya?l?k\u00e7? Y. C. Tu?rul L. Orosa H. Luo M. Patel O. Ergin and O. Mutlu. 2023. DRAM Bender: An Extensible and Versatile FPGA-Based Infrastructure to Easily Test State-of-the-Art DRAM Chips. IEEE TCAD (Dec. 2023).","DOI":"10.1109\/TCAD.2023.3282172"},{"key":"e_1_3_2_1_100_1","doi-asserted-by":"crossref","unstructured":"A. Olgun M. Patel A. G. Ya\u011flik\u00e7i H. Luo J. S. Kim F. N. Bostanci N. Vijaykumar O. Ergin and O. Mutlu. 2021. QUAC-TRNG: High-Throughput True Random Number Generation Using Quadruple Row Activation in Commodity DRAM Chips. In ISCA.","DOI":"10.1109\/ISCA52012.2021.00078"},{"key":"e_1_3_2_1_101_1","volume-title":"DAMOV: A New Methodology and Benchmark Suite for Evaluating Data Movement Bottlenecks","author":"Oliveira G. F.","year":"2021","unstructured":"G. F. Oliveira, J. G\u00f3mez-Luna, L. Orosa, S. Ghose, N. Vijaykumar, I. Fernandez, M. Sadrosadati, and O. Mutlu. 2021. DAMOV: A New Methodology and Benchmark Suite for Evaluating Data Movement Bottlenecks. IEEE Access (2021)."},{"key":"e_1_3_2_1_102_1","volume-title":"MIMDRAM: An End-to-End Processing-Using-DRAM System for High-Throughput, Energy-Efficient and Programmer-Transparent Multiple-Instruction Multiple-Data Computing. In HPCA.","author":"Oliveira G. F.","year":"2024","unstructured":"G. F. Oliveira, A. Olgun, A. G. Ya\u011flik\u00e7i, F. N. Bostanci, J. G\u00f3mez-Luna, S. Ghose, and O. Mutlu. 2024. MIMDRAM: An End-to-End Processing-Using-DRAM System for High-Throughput, Energy-Efficient and Programmer-Transparent Multiple-Instruction Multiple-Data Computing. In HPCA."},{"key":"e_1_3_2_1_103_1","unstructured":"OpenSSL Software Foundation Inc. [n.d.]. OpenSSL. https:\/\/github.com\/openssl\/openssl"},{"key":"e_1_3_2_1_104_1","doi-asserted-by":"crossref","unstructured":"J. Park R. Azizi G. F. Oliveira M. Sadrosadati R. Nadig D. Novo J. G\u00f3mez-Luna M. Kim and O. Mutlu. 2022. Flash-Cosmos: In-Flash Bulk Bitwise Operations Using Inherent Computation Capability of NAND Flash Memory. In MICRO.","DOI":"10.1109\/MICRO56248.2022.00069"},{"key":"e_1_3_2_1_105_1","unstructured":"A. Paszke S. Gross F. Massa A. Lerer J. Bradbury G. Chanan T. Killeen Z. Lin N. Gimelshein L. Antiga A. Desmaison A. Kopf E. Yang Z. DeVito M. Raison A. Tejani S. Chilamkurthy B. Steiner L. Fang J. Bai and S. Chintala. 2019. PyTorch: An Imperative Style High-Performance Deep Learning Library. In NeurIPS."},{"key":"e_1_3_2_1_106_1","doi-asserted-by":"crossref","unstructured":"J. T. Pawlowski. 2011. Hybrid Memory Cube (HMC). In HCS.","DOI":"10.1109\/HOTCHIPS.2011.7477494"},{"key":"e_1_3_2_1_107_1","doi-asserted-by":"crossref","unstructured":"G. Pedretti C. E. Graves S. Serebryakov R. Mao X. Sheng M. Foltin C. Li and J. P. Strachan. 2021. Tree-Based Machine Learning Performed In-Memory With Memristive Analog CAM. Nat. Commun. (Oct. 2021).","DOI":"10.1038\/s41467-021-25873-0"},{"key":"e_1_3_2_1_108_1","doi-asserted-by":"crossref","unstructured":"G. Pedretti S. Serebryakov J. P. Strachan and C. E. Graves. 2022. A General Tree-Based Machine Learning Accelerator With Memristive Analog CAM. In ISCAS.","DOI":"10.1109\/ISCAS48785.2022.9937772"},{"key":"e_1_3_2_1_109_1","volume-title":"Mesoscale Meteorological Modeling","author":"Pielke R. A.","unstructured":"R. A. Pielke Sr., 2013. Mesoscale Meteorological Modeling (2nd ed.). Academic Press.","edition":"2"},{"key":"e_1_3_2_1_110_1","volume-title":"High Performance Numerical Computing for High Energy Physics: A New Challenge for Big Data Science. Adv. High Energy Phys. (Feb","author":"Pop F.","year":"2014","unstructured":"F. Pop. 2014. High Performance Numerical Computing for High Energy Physics: A New Challenge for Big Data Science. Adv. High Energy Phys. (Feb. 2014)."},{"key":"e_1_3_2_1_111_1","doi-asserted-by":"crossref","unstructured":"M. R. H. Rashed S. K. Jha and R. Ewetz. 2021. Hybrid Analog-Digital In-Memory Computing. In ICCAD.","DOI":"10.1109\/ICCAD51958.2021.9643526"},{"key":"e_1_3_2_1_112_1","doi-asserted-by":"crossref","unstructured":"V. J. Reddi C. Cheng D. Kanter P. Mattson G. Schmuelling C.-J. Wu B. Anderson M. Breughe M. Charlebois W. Chou R. Chukka C. Coleman S. Davis P. Deng G. Diamos J. Duke D. Fick J. S. Gardner I. Hubara S. Idgunji T. B. Jablin J. Jiao T. St. John P. Kanwar D. Lee J. Liao A. Lokhmotov F. Massa P. Meng P. Micikevicius C. Osborne G. Pekhimenko A. T. R. Rajan D. Sequeira A. Sirasao F. Sun H. Tang M. Thomson F. Wei E. Wu L. Xu K. Yamada B. Yu G. Yuan A. Zhong P. Zhang and Y. Zhou. 2020. MLPerf Inference Benchmark. In ISCA.","DOI":"10.1109\/ISCA45697.2020.00045"},{"key":"e_1_3_2_1_113_1","volume":"201","author":"Reinsel D.","unstructured":"D. Reinsel, J. Gantz, and J. Rydning. 2018. Data Age 2025: The Digitization of the World From Edge to Core. Technical Report. IDC.","journal-title":"J. Rydning."},{"key":"e_1_3_2_1_114_1","doi-asserted-by":"crossref","unstructured":"S. Rhyner H. Luo J. G\u00f3mez-Luna M. Sadrosadati J. Jiang A. Olgun H. Gupta C. Zhang and O. Mutlu. 2024. PIM-Opt: Demystifying Distributed Optimization Algorithms on a Real-World Processing-in-Memory System. In PACT.","DOI":"10.1145\/3656019.3676947"},{"key":"e_1_3_2_1_115_1","unstructured":"J. K. Rott. 2012. Intel\u00ae Advanced Encryption Standard Instructions (AES-NI). Technical Report. \u00a9 Intel Corporation. https:\/\/www.intel.com\/content\/www\/us\/en\/developer\/articles\/technical\/advanced-encryption-standard-instructions-aes-ni.html"},{"key":"e_1_3_2_1_116_1","doi-asserted-by":"crossref","unstructured":"K. Roy A. Jaiswal and P. Panda. 2019. Towards Spike-Based Machine Intelligence With Neuromorphic Computing. Nature (2019).","DOI":"10.1038\/s41586-019-1677-2"},{"key":"e_1_3_2_1_117_1","doi-asserted-by":"crossref","unstructured":"U. Saxena and K. Roy. 2023. Partial-Sum Quantization for Near ADC-Less Compute-In-Memory Accelerators. In ISLPED.","DOI":"10.1109\/ISLPED58423.2023.10244291"},{"key":"e_1_3_2_1_118_1","doi-asserted-by":"publisher","DOI":"10.7551\/mitpress\/4057.001.0001"},{"key":"e_1_3_2_1_119_1","doi-asserted-by":"crossref","unstructured":"V. Seshadri K. Hsieh A. Boroum D. Lee M. A. Kozuch O. Mutlu P. B. Gibbons and T. C. Mowry. 2015. Fast Bulk Bitwise AND and OR in DRAM. CAL (Jul. 2015).","DOI":"10.1109\/LCA.2015.2434872"},{"key":"e_1_3_2_1_120_1","doi-asserted-by":"crossref","unstructured":"V. Seshadri Y. Kim C. Fallin D. Lee R. Ausavarungnirun G. Pekhimenko Y. Luo O. Mutlu P. B. Gibbons M. A. Kozuch and T. C. Mowry. 2013. RowClone: Fast and Energy-Efficient In-DRAM Bulk Data Copy and Initialization. In MICRO.","DOI":"10.1145\/2540708.2540725"},{"key":"e_1_3_2_1_121_1","doi-asserted-by":"publisher","DOI":"10.1145\/3123939.3124544"},{"key":"e_1_3_2_1_122_1","volume-title":"ISAAC: A Convolutional Neural Network Accelerator With In-Situ Analog Arithmetic in Crossbars. In ISCA.","author":"Shafiee A.","year":"2016","unstructured":"A. Shafiee, A. Nag, N. Muralimanohar, R. Balasubramonian, J. P. Strachan, M. Hu, R. S. Williams, and V. Srikumar. 2016. ISAAC: A Convolutional Neural Network Accelerator With In-Situ Analog Arithmetic in Crossbars. In ISCA."},{"key":"e_1_3_2_1_123_1","doi-asserted-by":"crossref","unstructured":"M. Sharad D. Fan and K. Roy. 2013. Ultra Low Power Associative Computing With Spin Neurons and Resistive Crossbar Memory. In DAC.","DOI":"10.1145\/2463209.2488866"},{"key":"e_1_3_2_1_124_1","doi-asserted-by":"crossref","unstructured":"W. Shim and S. Yu. 2022. GP3D: 3D NAND Based In-Memory Graph Processing Accelerator. JETCAS (Jun. 2022).","DOI":"10.1109\/JETCAS.2022.3155654"},{"key":"e_1_3_2_1_125_1","doi-asserted-by":"crossref","unstructured":"C. E. Song P. Bhatnagar Z. Xia N. S. Kim T. S. Rosing and M. Kang. 2025. Hybrid SLC-MLC RRAM Mixed-Signal Processing-in-Memory Architecture for Transformer Acceleration via Gradient Redistribution. In ISCA.","DOI":"10.1145\/3695053.3731109"},{"key":"e_1_3_2_1_126_1","doi-asserted-by":"crossref","unstructured":"L. Song F. Chen H. Li and Y. Chen. 2023. ReFloat: Low-Cost Floating-Point Processing in ReRAM for Accelerating Iterative Linear Solvers. In SC.","DOI":"10.1145\/3581784.3607077"},{"key":"e_1_3_2_1_127_1","doi-asserted-by":"crossref","unstructured":"L. Song X. Qian H. Li and Y. Chen. 2017. PipeLayer: A Pipelined ReRAM-Based Accelerator for Deep Learning. In HPCA.","DOI":"10.1109\/HPCA.2017.55"},{"key":"e_1_3_2_1_128_1","doi-asserted-by":"crossref","unstructured":"L. Song Y. Zhuo X. Qian H. Li and Y. Chen. 2018. GraphR: Accelerating Graph Processing Using ReRAM. In HPCA.","DOI":"10.1109\/HPCA.2018.00052"},{"key":"e_1_3_2_1_129_1","volume":"202","author":"Song W.","unstructured":"W. Song, M. Rao, Y. Li, C. Li, Y. Zhou, F. Cai, M. Wu, W. Yin, Z. Li, Q. Wei, S. Lee, H. Zhu, L. Gong, M. Barnell, Q. Wu, P. A. Beerel, M. S.-W. Chen, N. Ge, M. Hu, Q. Xia, and J. J. Yang. 2024. Programming Memristor Arrays With Arbitrarily High Precision for Analog Computing. Science (Feb. 2024).","journal-title":"J. Yang."},{"key":"e_1_3_2_1_130_1","doi-asserted-by":"crossref","unstructured":"M. Spear J. E. Kim C. H. Bennett S. Agarwal M. J. Marinella and T. P. Xiao. 2023. The Impact of Analog-to-Digital Converter Architecture and Variability on Analog Neural Network Accuracy. JXCDC (2023).","DOI":"10.1109\/JXCDC.2023.3315134"},{"key":"e_1_3_2_1_131_1","doi-asserted-by":"crossref","unstructured":"A. Stillmaker and B. Baas. 2017. Scaling Equations for the Accurate Prediction of CMOS Device Performance From 180nm to 7nm. Integration (2017).","DOI":"10.1016\/j.vlsi.2017.02.002"},{"key":"e_1_3_2_1_132_1","volume-title":"A Proposal for Toeplitz Matrix Calculations. Stud. Appl. Math. (Apr","author":"Strang G.","year":"1986","unstructured":"G. Strang. 1986. A Proposal for Toeplitz Matrix Calculations. Stud. Appl. Math. (Apr. 1986)."},{"key":"e_1_3_2_1_133_1","doi-asserted-by":"crossref","unstructured":"Y. Sun Y. Wang and H. Yang. 2017. Energy-Efficient SQL Query Exploiting RRAM-Based Process-in-Memory Structure. In NVMSA.","DOI":"10.1109\/NVMSA.2017.8064463"},{"key":"e_1_3_2_1_134_1","unstructured":"M. O. Topal A. Bas and I. van Heerden. 2021. Exploring Transformers in Natural Language Generation: GPT BERT and XLNet. arXiv:2102.08036 [cs.CL]."},{"key":"e_1_3_2_1_135_1","unstructured":"H. Touvron T. Lavril G. Izacard X. Martinet M.-A. Lachaux T. Lacroix B. Rozi\u00e8re N. Goyal E. Hambro F. Azhar A. Rodriguez A. Joulin E. Grave and G. Lample. 2023. LLaMA: Open and Efficient Foundation Language Models. arXiv preprint arXiv:2302.13971 (2023)."},{"key":"e_1_3_2_1_136_1","doi-asserted-by":"crossref","unstructured":"M.S.Q. Truong Y. Sun D. Xiong A. Shah A. Glass A. Farrell J. A. Bain L. R. Carley and S. Ghose. 2026. The Memory Processing Unit: A Generalized Interface for End-to-End In-Memory Execution. In HPCA.","DOI":"10.1109\/HPCA68181.2026.11408599"},{"key":"e_1_3_2_1_137_1","volume-title":"RACER: Bit-Pipelined Processing Using Resistive Memory. In MICRO.","author":"Truong M. S. Q.","year":"2021","unstructured":"M. S. Q. Truong, E. Chen, D. Su, L. Shen, A. Glass, L. R. Carley, J. A. Bain, and S. Ghose. 2021. RACER: Bit-Pipelined Processing Using Resistive Memory. In MICRO."},{"key":"e_1_3_2_1_138_1","doi-asserted-by":"crossref","unstructured":"M. S. Q. Truong L. Shen A. Glass A. Hoffmann L. R. Carley J. A. Bain and S. Ghose. 2022. Adapting the RACER Architecture to Integrate Improved In-ReRAM Logic Primitives. JETCAS (2022).","DOI":"10.1109\/JETCAS.2022.3171765"},{"key":"e_1_3_2_1_139_1","doi-asserted-by":"crossref","unstructured":"P.-H. Tseng F.-M. Lee Y.-H. Lin L.-Y. Chen Y.-C. Li H.-W. Hu Y.-Y. Wang C.-C. Hsieh M.-H. Lee H.-L. Lung K.-Y. Hsieh K.-C. Wang and C.-Y. Lu. 2020. In-Memory-Searching Architecture Based on 3D-NAND Technology With Ultra-High Parallelism. In IEDM.","DOI":"10.1109\/IEDM13553.2020.9372035"},{"key":"e_1_3_2_1_140_1","unstructured":"P.-H. Tseng F.-M. Lee Y.-H. Lin Y.-Y. Wang M.-H. Lee K.-Y. Hsieh K.-C. Wang and C.-Y. Lu. 2021. A Hybrid In-Memory-Searching and In-Memory-Computing Architecture for NVM Based AI Accelerator. In VLSI Symposia."},{"key":"e_1_3_2_1_141_1","unstructured":"A. Vaswani N. Shazeer N. Parmar J. Uszkoreit L. Jones A. N. Gomez \u0141. Kaiser and I. Polosukhin. 2017. Attention Is All You Need. NIPS (2017)."},{"key":"e_1_3_2_1_142_1","doi-asserted-by":"crossref","unstructured":"P. O. Vontobel W. Robinett P. J. Kuekes D. R. Stewart J. Straznicky and R. S. Williams. 2009. Writing to and Reading From a Nano-Scale Crossbar Memory Based on Memristors. Nanotechnology (Sep. 2009).","DOI":"10.1088\/0957-4484\/20\/42\/425204"},{"key":"e_1_3_2_1_143_1","doi-asserted-by":"crossref","unstructured":"J. Wang X. Wang C. Eckert A. Subramaniyan R. Das D. Blaauw and D. Sylvester. 2019. A Compute SRAM With Bit-Serial Integer\/Floating-Point Operations for Programmable In-Memory Vector Acceleration. In ISSCC.","DOI":"10.1109\/ISSCC.2019.8662419"},{"key":"e_1_3_2_1_144_1","doi-asserted-by":"crossref","unstructured":"J. Wang X. Wang C. Eckert A. Subramaniyan R. Das D. Blaauw and D. Sylvester. 2020. A 28-nm Compute SRAM With Bit-Serial Logic\/Arithmetic Operations for Programmable In-Memory Vector Computing. JSSC (Jan. 2020).","DOI":"10.1109\/JSSC.2019.2939682"},{"key":"e_1_3_2_1_145_1","doi-asserted-by":"crossref","unstructured":"J. Wang S. Yan X. Fu Z. Qian Z. Li Z. Guo Z. Dai Z. Cong C. Dou F. Zhang J. Yue and D. Shang. 2025b. A High-Density Energy-Efficient CNM Macro Using Hybrid RRAM and SRAM for Memory-Bound Applications. TVLSI (Jun. 2025).","DOI":"10.1109\/TVLSI.2025.3576889"},{"key":"e_1_3_2_1_146_1","doi-asserted-by":"crossref","unstructured":"S. Wang Z. Liu C. Ding C. Zhang T. Wu J. Zhou and G. Wong. 2025a. NoiseZO: RRAM Noise-Driven Zeroth-Order Optimization for Efficient Forward-Only Training. In DAC.","DOI":"10.1109\/DAC63849.2025.11132557"},{"key":"e_1_3_2_1_147_1","volume-title":"REREC: In-ReRAM Acceleration With Access-Aware Mapping for Personalized Recommendation. In ICCAD.","author":"Wang Y.","year":"2021","unstructured":"Y. Wang, Z. Zhu, F. Chen, M. Ma, G. Dai, Y. Wang, H. Li, and Y. Chen. 2021. REREC: In-ReRAM Acceleration With Access-Aware Mapping for Personalized Recommendation. In ICCAD."},{"key":"e_1_3_2_1_148_1","volume-title":"ANVIL: An In-Storage Accelerator for Name-Value Data Stores. In ISCA.","author":"Wong R.","year":"2025","unstructured":"R. Wong, N. Kim, A. Das, K. Higgs, E. Ipek, S. Agarwal, S. Ghose, and B. Feinberg. 2025. ANVIL: An In-Storage Accelerator for Name-Value Data Stores. In ISCA."},{"key":"e_1_3_2_1_149_1","volume":"202","author":"Xiao T. P.","unstructured":"T. P. Xiao, C. H. Bennett, B. Feinberg, S. Agarwal, and M. J. Marinella. 2020. Analog Architectures for Neural Network Acceleration Based on Non-Volatile Memory. APR (Jul. 2020).","journal-title":"J. Marinella."},{"key":"e_1_3_2_1_150_1","volume":"202","author":"Xiao T. P.","unstructured":"T. P. Xiao, B. Feinberg, C. H. Bennett, V. Agrawal, P. Saxena, V. Prabhakar, K. Ramkumar, H. Medu, V. Raghavan, R. Chettuvetty, S. Agarwal, and M. J. Marinella. 2022. An Accurate, Error-Tolerant, and Energy-Efficient Neural Network Inference Engine Based on SONOS Analog Memory. TCAS-I (Jan. 2022).","journal-title":"J. Marinella."},{"key":"e_1_3_2_1_151_1","volume":"202","author":"Xiao T. P.","unstructured":"T. P. Xiao, B. Feinberg, C. H. Bennett, V. Prabhakar, P. Saxena, V. Agrawal, S. Agarwal, and M. J. Marinella. 2023. On the Accuracy of Analog Neural Network Inference Accelerators. CAS (Jan. 2023).","journal-title":"J. Marinella."},{"key":"e_1_3_2_1_152_1","unstructured":"T. P. Xiao B. Feinberg D. K. Richardson M. Cannon H. Medu V. Agrawal M. J. Marinella S. Agarwal and C. H. Bennett. 2024. Analog Fast Fourier Transforms for Scalable and Efficient Signal Processing. arXiv:2409.19071 [cs.ET]."},{"key":"e_1_3_2_1_153_1","volume":"202","author":"Xiao T. P.","unstructured":"T. P. Xiao, B. Feinberg, J. N. Rohan, C. H. Bennett, S. Agarwal, and M. J. Marinella. 2021. Analysis and Mitigation of Parasitic Resistance Effects for Analog In-Memory Neural Network Acceleration. SST, Vol. 36 (Oct. 2021).","journal-title":"J. Marinella."},{"key":"e_1_3_2_1_154_1","volume-title":"AIM: Fast and Energy-Efficient AES In-Memory Implementation for Emerging Non-volatile Main Memory. In DATE.","author":"Xie M.","year":"2018","unstructured":"M. Xie, S. Li, A. O. Glova, J. Hu, Y. Wang, and Y. Xie. 2018. AIM: Fast and Energy-Efficient AES In-Memory Implementation for Emerging Non-volatile Main Memory. In DATE."},{"key":"e_1_3_2_1_155_1","doi-asserted-by":"crossref","unstructured":"C. Xu D. Niu N. Muralimanohar R. Balasubramonian T. Zhang S. Yu and Y. Xie. 2015. Overcoming the Challenges of Crossbar Resistive Memory Architectures. In HPCA.","DOI":"10.1109\/HPCA.2015.7056056"},{"key":"e_1_3_2_1_156_1","volume-title":"HEIRS: Hybrid Three-Dimension RRAM- and SRAM-CIM Architecture for Multi-task Transformer Acceleration. In DAC.","author":"Xu L.","year":"2024","unstructured":"L. Xu, S. Yuan, D. Wang, Y. Chen, X. Li, and Y. Sun. 2024. HEIRS: Hybrid Three-Dimension RRAM- and SRAM-CIM Architecture for Multi-task Transformer Acceleration. In DAC."},{"key":"e_1_3_2_1_157_1","doi-asserted-by":"crossref","unstructured":"B. Yan J.-L. Hsu P.-C. Yu C.-C. Lee Y. Zhang W. Yue G. Mei Y. Yang Y. Yang H. Li Y. Chen and R. Huang. 2022. A 1.041-Mb\/mm2 27.38-TOPS\/W Signed-INT8 Dynamic-Logic-Based ADC-less SRAM Compute-in-Memory Macro in 28nm With Reconfigurable Bitwise Operation for AI and Embedded Applications. In ISSCC.","DOI":"10.1109\/ISSCC42614.2022.9731545"},{"key":"e_1_3_2_1_158_1","doi-asserted-by":"crossref","unstructured":"B. Yan Z. Li Y. Chen and H. Li. 2016. RAM and TCAM Designs by Using STT-MRAM. In NVMTS.","DOI":"10.1109\/NVMTS.2016.7781514"},{"key":"e_1_3_2_1_159_1","doi-asserted-by":"crossref","unstructured":"T.-H. Yang H.-Y. Cheng C.-L. Yang I-C. Tseng H.-W. Hu H.-S. Chang and H.-P. Li. 2019. Sparse ReRAM Engine: Joint Exploration of Activation and Weight Sparsity in Compressed Neural Networks. In ISCA.","DOI":"10.1145\/3307650.3322271"},{"key":"e_1_3_2_1_160_1","volume-title":"Fully Hardware-Implemented Memristor Convolutional Neural Network. Nature","volume":"577","author":"Yao P.","year":"2020","unstructured":"P. Yao, H. Wu, B. Gao, J. Tang, Q. Zhang, W. Zhang, J. J. Yang, and H. Qian. 2020. Fully Hardware-Implemented Memristor Convolutional Neural Network. Nature, Vol. 577 (Jan. 2020)."},{"key":"e_1_3_2_1_161_1","doi-asserted-by":"crossref","unstructured":"A. Yazdanbakhsh A. Moradifirouzabadi Z. Li and M. Kang. 2022. Sparse Attention Acceleration With Synergistic In-Memory Pruning and On-Chip Recomputation. In MICRO.","DOI":"10.1109\/MICRO56248.2022.00059"},{"key":"e_1_3_2_1_162_1","volume-title":"FORMS: Fine-Grained Polarized ReRAM-Based In-Situ Computation for Mixed-Signal DNN Accelerator. In ISCA.","author":"Yuan G.","year":"2021","unstructured":"G. Yuan, P. Behnam, Z. Li, A. Shafiee, S. Lin, X. Ma, H. Liu, X. Qian, M. N. Bojnordi, Y. Wang, and C. Ding. 2021. FORMS: Fine-Grained Polarized ReRAM-Based In-Situ Computation for Mixed-Signal DNN Accelerator. In ISCA."},{"key":"e_1_3_2_1_163_1","doi-asserted-by":"crossref","unstructured":"?. E. Y\u00fcksel Y. C. Tu?rul F. N. Bostanc? G. F. Oliveira A. G. Ya?l?k\u00e7? A. Olgun M. Soysal H. Luo J. G\u00f3mez-Luna M. Sadrosadati and O. Mutlu. 2024a. Simultaneous Many-Row Activation in Off-the-Shelf DRAM Chips: Experimental Characterization and Analysis. In DSN.","DOI":"10.1109\/DSN58291.2024.00024"},{"key":"e_1_3_2_1_164_1","doi-asserted-by":"crossref","unstructured":"?. E. Y\u00fcksel Y. C. Tu?rul A. Olgun F. N. Bostanc? A. G. Ya?l?k\u00e7? G. F. Oliveira H. Luo J. G\u00f3mez-Luna M. Sadrosadati and O. Mutlu. 2024b. Functionally-Complete Boolean Logic in Real DRAM Chips: Experimental Characterization and Analysis. In HPCA.","DOI":"10.1109\/HPCA57654.2024.00030"},{"key":"e_1_3_2_1_165_1","volume":"2018","author":"Zha Y.","unstructured":"Y. Zha and J. Li. 2018a. Liquid Silicon: A Data-Centric Reconfigurable Architecture Enabled by RRAM Technology. In FPGA.","journal-title":"J. Li."},{"key":"e_1_3_2_1_166_1","volume":"2018","author":"Zha Y.","unstructured":"Y. Zha and J. Li. 2018b. Liquid Silicon-Monona: A Reconfigurable Memory-Oriented Computing Fabric With Scalable Multi-Context Support. In ASPLOS.","journal-title":"J. Li."},{"key":"e_1_3_2_1_167_1","doi-asserted-by":"crossref","unstructured":"F. Zhang A. Sridharan W. Tsai Y. Chen S. X. Wang and D. Fan. 2024a. Efficient Memory Integration: MRAM-SRAM Hybrid Accelerator for Sparse On-Device Learning. In DAC.","DOI":"10.1145\/3649329.3657390"},{"key":"e_1_3_2_1_168_1","doi-asserted-by":"crossref","unstructured":"F. Zhang L. Yang and D. Fan. 2024b. Hyb-Learn: A Framework for On-Device Self-Supervised Continual Learning With Hybrid RRAM\/SRAM Memory. In DAC.","DOI":"10.1145\/3649329.3657389"},{"key":"e_1_3_2_1_169_1","unstructured":"L. Zhao L. Buonanno A. Gajjar J. Moon A. Natarajan S. Serebryakov R. M. Roth X. Sheng Y. Zhang P. Faraboschi J. Ignowski and G. Pedretti. 2025a. NL-DPE: An Analog In-Memory Non-Linear Dot Product Engine for Efficient CNN and LLM Inference. arXiv:2511.13950 [cs.AR]."},{"key":"e_1_3_2_1_170_1","doi-asserted-by":"crossref","unstructured":"L. Zhao L. Buonanno R. M. Roth S. Serebryakov A. Gajjar J. Moon J. Ignowski and G. Pedretti. 2025b. RACE-IT: A Reconfigurable Analog CAM-Crossbar Engine for In-Memory Transformer Acceleration. In ICCD.","DOI":"10.1109\/ICCD65941.2025.00021"},{"key":"e_1_3_2_1_171_1","doi-asserted-by":"crossref","unstructured":"P. Zuo Q. Wang Y. Luo R. Xie S. Wang Z. Cheng L. Bao Z. Wang Y. Cai R. Huang and Z. Sun. 2025. Precise and Scalable Analogue Matrix Equation Solving Using Resistive Random-Access Memory Chips. Nat. Electron. (Oct. 2025).","DOI":"10.1038\/s41928-025-01477-0"}],"event":{"name":"ASPLOS '26: 31st ACM International Conference on Architectural Support for Programming Languages and Operating Systems","location":"Pittsburgh PA USA","sponsor":["SIGOPS ACM Special Interest Group on Operating Systems","SIGPLAN ACM Special Interest Group on Programming Languages","SIGARCH ACM Special Interest Group on Computer Architecture","SIGBED ACM Special Interest Group on Embedded Systems"]},"container-title":["Proceedings of the 31st ACM International Conference on Architectural Support for Programming Languages and Operating Systems, Volume 2"],"original-title":[],"deposited":{"date-parts":[[2026,3,15]],"date-time":"2026-03-15T14:06:59Z","timestamp":1773583619000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3779212.3790151"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,3,22]]},"references-count":171,"alternative-id":["10.1145\/3779212.3790151","10.1145\/3779212"],"URL":"https:\/\/doi.org\/10.1145\/3779212.3790151","relation":{},"subject":[],"published":{"date-parts":[[2026,3,22]]},"assertion":[{"value":"2026-03-22","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}