{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,11]],"date-time":"2026-07-11T16:54:25Z","timestamp":1783788865406,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":49,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,4,27]],"date-time":"2024-04-27T00:00:00Z","timestamp":1714176000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,4,27]]},"DOI":"10.1145\/3620666.3651353","type":"proceedings-article","created":{"date-parts":[[2024,4,24]],"date-time":"2024-04-24T12:08:21Z","timestamp":1713960501000},"page":"464-478","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":11,"title":["GMT: GPU Orchestrated Memory Tiering for the Big Data Era"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0002-6711-0071","authenticated-orcid":false,"given":"Chia-Hao","family":"Chang","sequence":"first","affiliation":[{"name":"The Pennsylvania State University, University Park, Pennsylvania, United States of America"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-0496-1462","authenticated-orcid":false,"given":"Jihoon","family":"Han","sequence":"additional","affiliation":[{"name":"The Pennsylvania State University, University Park, Pennsylvania, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6173-687X","authenticated-orcid":false,"given":"Anand","family":"Sivasubramaniam","sequence":"additional","affiliation":[{"name":"The Pennsylvania State University, University Park, Pennsylvania, United States of America"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9611-8075","authenticated-orcid":false,"given":"Vikram","family":"Sharma Mailthody","sequence":"additional","affiliation":[{"name":"NVIDIA Research, Santa Clara, United States of America"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1766-1289","authenticated-orcid":false,"given":"Zaid","family":"Qureshi","sequence":"additional","affiliation":[{"name":"NVIDIA Research, Santa Clara, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2532-5349","authenticated-orcid":false,"given":"Wen-Mei","family":"Hwu","sequence":"additional","affiliation":[{"name":"NVIDIA Research, Santa Clara, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2024,4,27]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"Control groups. https:\/\/docs.kernel.org\/admin-guide\/cgroup-v1\/cgroups.html."},{"key":"e_1_3_2_1_2_1","unstructured":"Gpudirect storage: A direct path between storage and gpu memory. https:\/\/developer.nvidia.com\/blog\/gpudirect-storage."},{"key":"e_1_3_2_1_3_1","unstructured":"Heterogeneous memory management. https:\/\/www.kernel.org\/doc\/html\/v5.0\/vm\/hmm.html."},{"key":"e_1_3_2_1_4_1","unstructured":"Numa balancing. https:\/\/mirrors.edge.kernel.org."},{"key":"e_1_3_2_1_5_1","unstructured":"Simplifying gpu application development with heterogeneous memory management. https:\/\/developer.nvidia.com\/blog\/simplifying-gpu-application-development-with-heterogeneous-memory-management\/."},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1145\/3037697.3037706"},{"key":"e_1_3_2_1_7_1","volume-title":"The gap benchmark suite","author":"Beamer Scott","year":"2017","unstructured":"Scott Beamer, Krste Asanovi\u0107, and David Patterson. The gap benchmark suite, 2017."},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1147\/sj.52.0078"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1145\/3456727.3463766"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1145\/2370816.2370860"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1109\/IISWC.2009.5306797"},{"key":"e_1_3_2_1_12_1","unstructured":"Nikolay Sakharnykh Chirayu Garg. Improving gpu memory oversubscription performance. https:\/\/developer.nvidia.com\/blog\/improving-gpu-memory-oversubscription-performance\/."},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS.2012.100"},{"key":"e_1_3_2_1_14_1","volume-title":"Reuse distance-based probabilistic cache replacement. ACM Trans. Archit. Code Optim., 12(4), oct","author":"Das Subhasis","year":"2015","unstructured":"Subhasis Das, Tor M. Aamodt, and William J. Dally. Reuse distance-based probabilistic cache replacement. ACM Trans. Archit. Code Optim., 12(4), oct 2015."},{"key":"e_1_3_2_1_15_1","volume-title":"Davis and Yifan Hu. The university of florida sparse matrix collection. ACM Trans. Math. Softw., 38(1), dec","author":"Timothy","year":"2011","unstructured":"Timothy A. Davis and Yifan Hu. The university of florida sparse matrix collection. ACM Trans. Math. Softw., 38(1), dec 2011."},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1145\/780822.781159"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/MICRO.2012.43"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1109\/MICRO.2012.43"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1007\/11859802_6"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/PACT.2017.32"},{"key":"e_1_3_2_1_22_1","volume-title":"JWAC 2010 - 1st JILP Worshop on Computer Architecture Competitions: cache replacement Championship","author":"Gao Hongliang","year":"2010","unstructured":"Hongliang Gao and Chris Wilkerson. A Dueling Segmented LRU Replacement Algorithm with Adaptive Bypassing. In Joel Emer, editor, JWAC 2010 - 1st JILP Worshop on Computer Architecture Competitions: cache replacement Championship, Saint Malo, France, June 2010."},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1145\/2000064.2000075"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1145\/1816038.1815971"},{"key":"e_1_3_2_1_25_1","first-page":"436","volume-title":"2017 50th Annual IEEE\/ACM International Symposium on Microarchitecture (MICRO)","author":"Daniel","year":"2017","unstructured":"Daniel A. Jim\u00e9nez and Elvira Teran. Multiperspective reuse prediction. In 2017 50th Annual IEEE\/ACM International Symposium on Microarchitecture (MICRO), pages 436--448, 2017."},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1109\/12.817393"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1145\/264107.264213"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1145\/3140659.3080245"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCD.2005.41"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1109\/SC.2018.00035"},{"key":"e_1_3_2_1_32_1","unstructured":"Jonas Markusse. libnvm: An api for building userspace nvme drivers and storage applications. https:\/\/github.com\/enfiskutensykkel\/ssdgpu-dma."},{"key":"e_1_3_2_1_33_1","volume-title":"P\u00e5l Halvorsen, Halvor Kielland-Gyrud, H\u00e5kon Kvale Stensland, and Carsten Griwodz. Smartio: Zero-overhead device sharing through pcie networking. ACM Transactions on Computer Systems, 38(1--2), jul","author":"Markussen Jonas","year":"2021","unstructured":"Jonas Markussen, Lars Bj\u00f8rlykke Kristiansen, P\u00e5l Halvorsen, Halvor Kielland-Gyrud, H\u00e5kon Kvale Stensland, and Carsten Griwodz. Smartio: Zero-overhead device sharing through pcie networking. ACM Transactions on Computer Systems, 38(1--2), jul 2021."},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1145\/3582016.3582063"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.14778\/3425879.3425883"},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1016\/B978-0-12-801913-9.00003-8"},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1145\/149439.133084"},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.5555\/1812707.1812719"},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1145\/3575693.3575748"},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1145\/3477132.3483550"},{"key":"e_1_3_2_1_42_1","volume-title":"Mandana Bagheri Marzijarani, and Didem Unat. Reusetracker: Fast yet accurate multicore reuse distance analyzer. ACM Trans. Archit. Code Optim., 19(1), dec","author":"Sasongko Muhammad Aditya","year":"2021","unstructured":"Muhammad Aditya Sasongko, Milind Chabbi, Mandana Bagheri Marzijarani, and Didem Unat. Reusetracker: Fast yet accurate multicore reuse distance analyzer. ACM Trans. Archit. Code Optim., 19(1), dec 2021."},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA53966.2022.00048"},{"key":"e_1_3_2_1_44_1","first-page":"596","volume-title":"Proceedings of the 43rd International Symposium on Computer Architecture, ISCA '16","author":"Shahar Sagi","year":"2016","unstructured":"Sagi Shahar, Shai Bergman, and Mark Silberstein. Activepointers: A case for software address translation on gpus. In Proceedings of the 43rd International Symposium on Computer Architecture, ISCA '16, page 596--608. IEEE Press, 2016."},{"key":"e_1_3_2_1_45_1","volume-title":"Gpufs: Integrating a file system with gpus. ACM Trans. Comput. Syst., 32(1), feb","author":"Silberstein Mark","year":"2014","unstructured":"Mark Silberstein, Bryan Ford, Idit Keidar, and Emmett Witchel. Gpufs: Integrating a file system with gpus. ACM Trans. Comput. Syst., 32(1), feb 2014."},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.1109\/12.811113"},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA.2019.00056"},{"key":"e_1_3_2_1_48_1","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA56546.2023.10071033"},{"key":"e_1_3_2_1_49_1","first-page":"430","volume-title":"2011 44th Annual IEEE\/ACM International Symposium on Microarchitecture (MICRO)","author":"Wu Carole-Jean","year":"2011","unstructured":"Carole-Jean Wu, Aamer Jaleel, Will Hasenplaugh, Margaret Martonosi, Simon C. Steely, and Joel Emer. Ship: Signature-based hit predictor for high performance caching. In 2011 44th Annual IEEE\/ACM International Symposium on Microarchitecture (MICRO), pages 430--441, 2011."},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"publisher","DOI":"10.1145\/1542275.1542290"},{"key":"e_1_3_2_1_51_1","doi-asserted-by":"publisher","DOI":"10.1145\/3297858.3304024"},{"key":"e_1_3_2_1_52_1","doi-asserted-by":"publisher","DOI":"10.1145\/3613424.3614309"}],"event":{"name":"ASPLOS '24: 29th ACM International Conference on Architectural Support for Programming Languages and Operating Systems, Volume 3","location":"La Jolla CA USA","acronym":"ASPLOS '24","sponsor":["SIGARCH ACM Special Interest Group on Computer Architecture","SIGOPS ACM Special Interest Group on Operating Systems","SIGPLAN ACM Special Interest Group on Programming Languages","SIGBED ACM Special Interest Group on Embedded Systems"]},"container-title":["Proceedings of the 29th ACM International Conference on Architectural Support for Programming Languages and Operating Systems, Volume 3"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3620666.3651353","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T00:03:43Z","timestamp":1750291423000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3620666.3651353"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,4,27]]},"references-count":49,"alternative-id":["10.1145\/3620666.3651353","10.1145\/3620666"],"URL":"https:\/\/doi.org\/10.1145\/3620666.3651353","relation":{},"subject":[],"published":{"date-parts":[[2024,4,27]]},"assertion":[{"value":"2024-04-27","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}