{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,3]],"date-time":"2026-04-03T22:06:49Z","timestamp":1775254009119,"version":"3.50.1"},"publisher-location":"New York, NY, USA","reference-count":85,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,6,21]]},"DOI":"10.1145\/3695053.3731054","type":"proceedings-article","created":{"date-parts":[[2025,6,20]],"date-time":"2025-06-20T16:46:17Z","timestamp":1750437977000},"page":"601-615","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":1,"title":["Magellan: A High-Performance Loop-Guided Prefetcher for Indirect Memory Access"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-7331-0830","authenticated-orcid":false,"given":"Gelin","family":"Fu","sequence":"first","affiliation":[{"name":"Xi'an Jiaotong University, Xi'an, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2520-3731","authenticated-orcid":false,"given":"Tian","family":"Xia","sequence":"additional","affiliation":[{"name":"Xi'an Jiaotong University, Xi'an, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-1495-5426","authenticated-orcid":false,"given":"Mingzhuo","family":"Yin","sequence":"additional","affiliation":[{"name":"Xi'an Jiaotong University, Xi'an, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1732-4314","authenticated-orcid":false,"given":"Prashant J.","family":"Nair","sequence":"additional","affiliation":[{"name":"The University of British Columbia, Vancouver, Canada"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3376-4384","authenticated-orcid":false,"given":"Mieszko","family":"Lis","sequence":"additional","affiliation":[{"name":"The University of British Columbia, Vancouver, Canada"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1163-2014","authenticated-orcid":false,"given":"Pengju","family":"Ren","sequence":"additional","affiliation":[{"name":"Xi'an Jiaotong University, Xi'an, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,6,20]]},"reference":[{"key":"e_1_3_3_1_2_2","doi-asserted-by":"publisher","DOI":"10.1145\/2749469.2750386"},{"key":"e_1_3_3_1_3_2","doi-asserted-by":"publisher","DOI":"10.1145\/2925426.2926254"},{"key":"e_1_3_3_1_4_2","doi-asserted-by":"crossref","unstructured":"Sam Ainsworth and Timothy\u00a0M Jones. 2018. An event-triggered programmable prefetcher for irregular workloads. ACM Sigplan Notices 53 2 (2018) 578\u2013592.","DOI":"10.1145\/3296957.3173189"},{"key":"e_1_3_3_1_5_2","doi-asserted-by":"crossref","unstructured":"Sam Ainsworth and Timothy\u00a0M Jones. 2019. Software prefetching for indirect memory accesses: A microarchitectural perspective. ACM Transactions on Computer Systems (TOCS) 36 3 (2019) 1\u201334.","DOI":"10.1145\/3319393"},{"key":"e_1_3_3_1_6_2","doi-asserted-by":"publisher","DOI":"10.1145\/3373376.3378498"},{"key":"e_1_3_3_1_7_2","doi-asserted-by":"publisher","DOI":"10.1145\/125826.125932"},{"key":"e_1_3_3_1_8_2","doi-asserted-by":"crossref","unstructured":"David\u00a0H Bailey Eric Barszcz Leonardo Dagum and Horst\u00a0D Simon. 1993. NAS parallel benchmark results. IEEE Parallel & Distributed Technology: Systems & Applications 1 1 (1993) 43\u201351.","DOI":"10.1109\/88.219861"},{"key":"e_1_3_3_1_9_2","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA.2018.00021"},{"key":"e_1_3_3_1_10_2","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA.2019.00053"},{"key":"e_1_3_3_1_11_2","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA51647.2021.00062"},{"key":"e_1_3_3_1_12_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICDE.2013.6544839"},{"key":"e_1_3_3_1_13_2","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA.2019.00051"},{"key":"e_1_3_3_1_14_2","unstructured":"Scott Beamer Krste Asanovi\u0107 and David Patterson. 2015. The GAP benchmark suite. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1508.03619 (2015)."},{"key":"e_1_3_3_1_15_2","doi-asserted-by":"publisher","DOI":"10.1145\/3352460.3358325"},{"key":"e_1_3_3_1_16_2","doi-asserted-by":"crossref","unstructured":"Nathan Binkert Bradford Beckmann Gabriel Black Steven\u00a0K Reinhardt Ali Saidi Arkaprava Basu Joel Hestness Derek\u00a0R Hower Tushar Krishna Somayeh Sardashti et\u00a0al. 2011. The gem5 simulator. ACM SIGARCH computer architecture news 39 2 (2011) 1\u20137.","DOI":"10.1145\/2024716.2024718"},{"key":"e_1_3_3_1_17_2","doi-asserted-by":"publisher","DOI":"10.1145\/988672.988752"},{"key":"e_1_3_3_1_18_2","doi-asserted-by":"publisher","unstructured":"Adam\u00a0L. Buchsbaum Loukas Georgiadis Haim Kaplan Anne Rogers Robert\u00a0E. Tarjan and Jeffery\u00a0R. Westbrook. 2008. Linear-Time Algorithms for Dominators and Other Path-Evaluation Problems. SIAM J. Comput. 38 4 (Nov. 2008) 1533\u20131573. 10.1137\/070693217","DOI":"10.1137\/070693217"},{"key":"e_1_3_3_1_19_2","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS.2008.4536313"},{"key":"e_1_3_3_1_20_2","doi-asserted-by":"crossref","unstructured":"Shimin Chen Anastassia Ailamaki Phillip\u00a0B Gibbons and Todd\u00a0C Mowry. 2007. Improving hash join performance through prefetching. ACM Transactions on Database Systems (TODS) 32 3 (2007) 17\u2013es.","DOI":"10.1145\/1272743.1272747"},{"key":"e_1_3_3_1_21_2","first-page":"578","volume-title":"13th USENIX Symposium on Operating Systems Design and Implementation (OSDI 18)","author":"Chen Tianqi","year":"2018","unstructured":"Tianqi Chen, Thierry Moreau, Ziheng Jiang, Lianmin Zheng, Eddie Yan, Haichen Shen, Meghan Cowan, Leyuan Wang, Yuwei Hu, Luis Ceze, et\u00a0al. 2018. { TVM} : An automated { End-to-End} optimizing compiler for deep learning. In 13th USENIX Symposium on Operating Systems Design and Implementation (OSDI 18). 578\u2013594."},{"key":"e_1_3_3_1_22_2","doi-asserted-by":"crossref","unstructured":"Tien-Fu Chen and Jean-Loup Baer. 1995. Effective hardware-based data prefetching for high-performance processors. IEEE transactions on computers 44 5 (1995) 609\u2013623.","DOI":"10.1109\/12.381947"},{"key":"e_1_3_3_1_23_2","doi-asserted-by":"publisher","DOI":"10.5555\/774861.774869"},{"key":"e_1_3_3_1_24_2","doi-asserted-by":"publisher","DOI":"10.1145\/379240.379248"},{"key":"e_1_3_3_1_25_2","doi-asserted-by":"crossref","unstructured":"Timothy\u00a0A Davis and Yifan Hu. 2011. The University of Florida sparse matrix collection. ACM Transactions on Mathematical Software (TOMS) 38 1 (2011) 1\u201325.","DOI":"10.1145\/2049662.2049663"},{"key":"e_1_3_3_1_26_2","doi-asserted-by":"crossref","unstructured":"Jack Dongarra Michael\u00a0A Heroux and Piotr Luszczek. 2015. HPCG benchmark: a new metric for ranking high performance computing systems. Knoxville Tennessee 42 (2015).","DOI":"10.1177\/1094342015593158"},{"key":"e_1_3_3_1_27_2","doi-asserted-by":"crossref","unstructured":"Jack Doweck Wen-Fu Kao Allen Kuan-yu Lu Julius Mandelblat Anirudha Rahatekar Lihu Rappoport Efraim Rotem Ahmad Yasin and Adi Yoaz. 2017. Inside 6th-generation intel core: New microarchitecture code-named skylake. IEEE Micro 37 2 (2017) 52\u201362.","DOI":"10.1109\/MM.2017.38"},{"key":"e_1_3_3_1_28_2","volume-title":"A primer on hardware prefetching","author":"Falsafi Babak","year":"2022","unstructured":"Babak Falsafi and Thomas\u00a0F Wenisch. 2022. A primer on hardware prefetching. Springer Nature."},{"key":"e_1_3_3_1_29_2","doi-asserted-by":"publisher","DOI":"10.1145\/3470496.3527395"},{"key":"e_1_3_3_1_30_2","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA57654.2024.00040"},{"key":"e_1_3_3_1_31_2","doi-asserted-by":"publisher","DOI":"10.1145\/2976749.2978356"},{"key":"e_1_3_3_1_32_2","doi-asserted-by":"publisher","DOI":"10.1145\/3243176.3243181"},{"key":"e_1_3_3_1_33_2","doi-asserted-by":"crossref","unstructured":"John\u00a0L Henning. 2006. SPEC CPU2006 benchmark descriptions. ACM SIGARCH Computer Architecture News 34 4 (2006) 1\u201317.","DOI":"10.1145\/1186736.1186737"},{"key":"e_1_3_3_1_34_2","doi-asserted-by":"publisher","DOI":"10.1109\/SP54263.2024.00158"},{"key":"e_1_3_3_1_35_2","unstructured":"Torsten Hoefler Dan Alistarh Tal Ben-Nun Nikoli Dryden and Alexandra Peste. 2021. Sparsity in deep learning: Pruning and growth for efficient inference and training in neural networks. Journal of Machine Learning Research 22 241 (2021) 1\u2013124."},{"key":"e_1_3_3_1_36_2","doi-asserted-by":"crossref","unstructured":"Tatsushi Inagaki Tamiya Onodera Hideaki Komatsu and Toshio Nakatani. 2003. Stride prefetching by dynamically inspecting objects. ACM SIGPLAN Notices 38 5 (2003) 269\u2013277.","DOI":"10.1145\/780822.781161"},{"key":"e_1_3_3_1_37_2","unstructured":"Intel\u00ae 64 and IA-32 Architectures Optimization Reference Manual. [n. d.]. https:\/\/www.intel.com\/content\/www\/us\/en\/content-details\/671488\/intel-64-and-ia-32-architectures-optimization-reference-manual.html."},{"key":"e_1_3_3_1_38_2","doi-asserted-by":"publisher","DOI":"10.1145\/2540708.2540730"},{"key":"e_1_3_3_1_39_2","doi-asserted-by":"publisher","DOI":"10.1145\/3492321.3519583"},{"key":"e_1_3_3_1_40_2","doi-asserted-by":"publisher","DOI":"10.1145\/2749469.2750392"},{"key":"e_1_3_3_1_41_2","first-page":"206","volume-title":"Proceedings Sixth International Symposium on High-Performance Computer Architecture. HPCA-6 (Cat. No. PR00550)","author":"Karlsson Magnus","year":"2000","unstructured":"Magnus Karlsson, Fredrik Dahlgren, and Per Stenstrom. 2000. A prefetching technique for irregular accesses to linked data structures. In Proceedings Sixth International Symposium on High-Performance Computer Architecture. HPCA-6 (Cat. No. PR00550). IEEE, 206\u2013217."},{"key":"e_1_3_3_1_42_2","doi-asserted-by":"publisher","DOI":"10.1109\/MICRO.2016.7783763"},{"key":"e_1_3_3_1_43_2","unstructured":"Thomas\u00a0N Kipf and Max Welling. 2016. Semi-supervised classification with graph convolutional networks. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1609.02907 (2016)."},{"key":"e_1_3_3_1_44_2","doi-asserted-by":"publisher","DOI":"10.1145\/2540708.2540748"},{"key":"e_1_3_3_1_45_2","doi-asserted-by":"publisher","DOI":"10.1109\/SP.2019.00002"},{"key":"e_1_3_3_1_46_2","doi-asserted-by":"publisher","DOI":"10.1109\/CGO.2004.1281665"},{"key":"e_1_3_3_1_47_2","doi-asserted-by":"crossref","unstructured":"Jure Leskovec and Rok Sosi\u010d. 2016. Snap: A general-purpose network analysis and graph-mining library. ACM Transactions on Intelligent Systems and Technology (TIST) 8 1 (2016) 1\u201320.","DOI":"10.1145\/2898361"},{"key":"e_1_3_3_1_48_2","first-page":"643","volume-title":"31st USENIX Security Symposium (USENIX Security 22)","author":"Lipp Moritz","year":"2022","unstructured":"Moritz Lipp, Daniel Gruss, and Michael Schwarz. 2022. { AMD} prefetch attacks through power and time. In 31st USENIX Security Symposium (USENIX Security 22). 643\u2013660."},{"key":"e_1_3_3_1_49_2","doi-asserted-by":"publisher","DOI":"10.1109\/SP.2015.43"},{"key":"e_1_3_3_1_50_2","doi-asserted-by":"publisher","DOI":"10.1145\/237090.237190"},{"key":"e_1_3_3_1_51_2","doi-asserted-by":"crossref","unstructured":"Andrew Lumsdaine Douglas Gregor Bruce Hendrickson and Jonathan Berry. 2007. Challenges in parallel graph processing. Parallel Processing Letters 17 01 (2007) 5\u201320.","DOI":"10.1142\/S0129626407002843"},{"key":"e_1_3_3_1_52_2","unstructured":"Piotr Luszczek Jack\u00a0J Dongarra David Koester Rolf Rabenseifner Bob Lucas Jeremy Kepner John McCalpin David Bailey and Daisuke Takahashi. 2005. Introduction to the HPC challenge benchmark suite. (2005)."},{"key":"e_1_3_3_1_53_2","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA.2016.7446087"},{"key":"e_1_3_3_1_54_2","doi-asserted-by":"publisher","DOI":"10.1145\/1298306.1298311"},{"key":"e_1_3_3_1_55_2","doi-asserted-by":"crossref","unstructured":"Todd\u00a0C Mowry Monica\u00a0S Lam and Anoop Gupta. 1992. Design and evaluation of a compiler algorithm for prefetching. ACM Sigplan Notices 27 9 (1992) 62\u201373.","DOI":"10.1145\/143371.143488"},{"key":"e_1_3_3_1_56_2","doi-asserted-by":"publisher","DOI":"10.1109\/MICRO.2018.00010"},{"key":"e_1_3_3_1_57_2","doi-asserted-by":"publisher","DOI":"10.1145\/3352460.3358254"},{"key":"e_1_3_3_1_58_2","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA.2003.1183532"},{"key":"e_1_3_3_1_59_2","doi-asserted-by":"publisher","DOI":"10.1145\/2807591.2807626"},{"key":"e_1_3_3_1_60_2","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA52012.2021.00024"},{"key":"e_1_3_3_1_61_2","doi-asserted-by":"publisher","DOI":"10.1145\/3613424.3614255"},{"key":"e_1_3_3_1_62_2","doi-asserted-by":"publisher","DOI":"10.1109\/MICRO56248.2022.00072"},{"key":"e_1_3_3_1_63_2","doi-asserted-by":"crossref","unstructured":"Nicholas Nethercote and Julian Seward. 2007. Valgrind: a framework for heavyweight dynamic binary instrumentation. ACM Sigplan notices 42 6 (2007) 89\u2013100.","DOI":"10.1145\/1273442.1250746"},{"key":"e_1_3_3_1_64_2","doi-asserted-by":"publisher","DOI":"10.1109\/MICRO50266.2020.00056"},{"key":"e_1_3_3_1_65_2","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA56546.2023.10071026"},{"key":"e_1_3_3_1_66_2","unstructured":"Intel\u00ae oneAPI Programming\u00a0Guide. 2024. https:\/\/www.intel.com\/content\/www\/us\/en\/developer\/tools\/oneapi\/toolkits.html."},{"key":"e_1_3_3_1_67_2","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA45697.2020.00021"},{"key":"e_1_3_3_1_68_2","doi-asserted-by":"publisher","DOI":"10.1109\/CGO.2019.8661201"},{"key":"e_1_3_3_1_69_2","doi-asserted-by":"publisher","DOI":"10.1145\/2749469.2749473"},{"key":"e_1_3_3_1_70_2","doi-asserted-by":"publisher","DOI":"10.1109\/MICRO50266.2020.00078"},{"key":"e_1_3_3_1_71_2","doi-asserted-by":"publisher","DOI":"10.1145\/3466752.3480126"},{"key":"e_1_3_3_1_72_2","doi-asserted-by":"publisher","DOI":"10.1145\/300979.300989"},{"key":"e_1_3_3_1_73_2","volume-title":"19th USENIX Security Symposium (USENIX Security 10)","author":"Sehr David","year":"2010","unstructured":"David Sehr, Robert Muth, Cliff Biffle, Victor Khimenko, Egor Pasko, Karl Schimpf, Bennet Yee, and Brad Chen. 2010. Adapting Software Fault Isolation to Contemporary { CPU} Architectures. In 19th USENIX Security Symposium (USENIX Security 10)."},{"key":"e_1_3_3_1_74_2","doi-asserted-by":"crossref","unstructured":"Alan\u00a0Jay Smith. 1978. Sequential program prefetching in memory hierarchies. Computer 11 12 (1978) 7\u201321.","DOI":"10.1109\/C-M.1978.218016"},{"key":"e_1_3_3_1_75_2","doi-asserted-by":"publisher","DOI":"10.1145\/237721.237727"},{"key":"e_1_3_3_1_76_2","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA51647.2021.00061"},{"key":"e_1_3_3_1_77_2","doi-asserted-by":"crossref","unstructured":"Gang Tan et\u00a0al. 2017. Principles and implementation techniques of software-based fault isolation. Foundations and Trends\u00ae in Privacy and Security 1 3 (2017) 137\u2013198.","DOI":"10.1561\/3300000013"},{"key":"e_1_3_3_1_78_2","doi-asserted-by":"publisher","DOI":"10.1145\/168619.168635"},{"key":"e_1_3_3_1_79_2","doi-asserted-by":"publisher","DOI":"10.1145\/3352460.3358300"},{"key":"e_1_3_3_1_80_2","doi-asserted-by":"publisher","DOI":"10.1145\/3307650.3322225"},{"key":"e_1_3_3_1_81_2","doi-asserted-by":"publisher","DOI":"10.1145\/512529.512555"},{"key":"e_1_3_3_1_82_2","doi-asserted-by":"crossref","unstructured":"Tian Xia Gelin Fu Chenyang Li Zhongpei Luo Lucheng Zhang Ruiyang Chen Wenzhe Zhao Nanning Zheng and Pengju Ren. 2022. A comprehensive performance model of sparse matrix-vector multiplication to guide kernel optimization. IEEE Transactions on Parallel and Distributed Systems 34 2 (2022) 519\u2013534.","DOI":"10.1109\/TPDS.2022.3225230"},{"key":"e_1_3_3_1_83_2","doi-asserted-by":"crossref","unstructured":"Feng Xue Chenji Han Xinyu Li Junliang Wu Tingting Zhang Tianyi Liu Yifan Hao Zidong Du Qi Guo and Fuxin Zhang. 2024. Tyche: An Efficient and General Prefetcher for Indirect Memory Accesses. ACM Transactions on Architecture and Code Optimization 21 2 (2024) 1\u201326.","DOI":"10.1145\/3641853"},{"key":"e_1_3_3_1_84_2","first-page":"719","volume-title":"23rd USENIX Security Symposium (USENIX Security 14)","author":"Yarom Yuval","year":"2014","unstructured":"Yuval Yarom and Katrina Falkner. 2014. FLUSH+RELOAD: A High Resolution, Low Noise, L3 Cache Side-Channel Attack. In 23rd USENIX Security Symposium (USENIX Security 14). USENIX Association, San Diego, CA, 719\u2013732. https:\/\/www.usenix.org\/conference\/usenixsecurity14\/technical-sessions\/presentation\/yarom"},{"key":"e_1_3_3_1_85_2","doi-asserted-by":"publisher","DOI":"10.1145\/2830772.2830807"},{"key":"e_1_3_3_1_86_2","doi-asserted-by":"publisher","DOI":"10.1145\/3620665.3640396"}],"event":{"name":"ISCA '25: Proceedings of the 52nd Annual International Symposium on Computer Architecture","location":"Tokyo Japan","acronym":"SIGARCH '25","sponsor":["SIGARCH ACM Special Interest Group on Computer Architecture"]},"container-title":["Proceedings of the 52nd Annual International Symposium on Computer Architecture"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3695053.3731054","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,21]],"date-time":"2025-06-21T11:05:37Z","timestamp":1750503937000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3695053.3731054"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,6,20]]},"references-count":85,"alternative-id":["10.1145\/3695053.3731054","10.1145\/3695053"],"URL":"https:\/\/doi.org\/10.1145\/3695053.3731054","relation":{},"subject":[],"published":{"date-parts":[[2025,6,20]]},"assertion":[{"value":"2025-06-20","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}