{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,15]],"date-time":"2026-03-15T15:30:08Z","timestamp":1773588608985,"version":"3.50.1"},"publisher-location":"New York, NY, USA","reference-count":52,"publisher":"ACM","funder":[{"name":"NSF &#x28;National Science Foundation&#x29;","award":["2211018"],"award-info":[{"award-number":["2211018"]}]},{"name":"NSF &#x28;National Science Foundation&#x29;","award":["1909004"],"award-info":[{"award-number":["1909004"]}]},{"name":"NSF &#x28;National Science Foundation&#x29;","award":["1763681"],"award-info":[{"award-number":["1763681"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,3,22]]},"DOI":"10.1145\/3779212.3790130","type":"proceedings-article","created":{"date-parts":[[2026,3,10]],"date-time":"2026-03-10T13:55:26Z","timestamp":1773150926000},"page":"208-222","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Asynchrony and GPUs: Bridging this Dichotomy for I\/O with AGIO"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0008-0496-1462","authenticated-orcid":false,"given":"Jihoon","family":"Han","sequence":"first","affiliation":[{"name":"The Pennsylvania State University, University Park, PA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6173-687X","authenticated-orcid":false,"given":"Anand","family":"Sivasubramaniam","sequence":"additional","affiliation":[{"name":"The Pennsylvania State University, University Park, PA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-6711-0071","authenticated-orcid":false,"given":"Chia-Hao","family":"Chang","sequence":"additional","affiliation":[{"name":"Nvidia, Santa Clara, CA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9611-8075","authenticated-orcid":false,"given":"Vikram Sharma","family":"Mailthody","sequence":"additional","affiliation":[{"name":"Nvidia Research, Santa Clara, CA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1766-1289","authenticated-orcid":false,"given":"Zaid","family":"Qureshi","sequence":"additional","affiliation":[{"name":"Nvidia Research, Santa Clara, CA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2532-5349","authenticated-orcid":false,"given":"Wen-Mei","family":"Hwu","sequence":"additional","affiliation":[{"name":"Nvidia Research, Santa Clara, CA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2026,3,22]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1145\/1572769.1572792"},{"key":"e_1_3_2_1_2_1","unstructured":"AMD. pty. Unified memory. https:\/\/rocm.docs.amd.com\/projects\/HIP\/en\/docs-6.2.0\/how-to\/unified_memory.html"},{"key":"e_1_3_2_1_3_1","unstructured":"Jens Axboe. 2019. Efficient IO with io_uring. https:\/\/kernel.dk\/io_uring.pdf"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1145\/2063384.2063400"},{"key":"e_1_3_2_1_5_1","unstructured":"Scott Beamer Krste Asanovi? and David Patterson. 2017. The GAP Benchmark Suite. arXiv:1508.03619 [cs.DC] https:\/\/arxiv.org\/abs\/1508.03619"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1145\/3178487.3178492"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1145\/3309987"},{"key":"e_1_3_2_1_8_1","first-page":"661","volume-title":"GAIA: An OS Page Cache for Heterogeneous Systems. In 2019 USENIX Annual Technical Conference (USENIX ATC 19)","author":"Brokhman Tanya","year":"2019","unstructured":"Tanya Brokhman, Pavel Lifshits, and Mark Silberstein. 2019. GAIA: An OS Page Cache for Heterogeneous Systems. In 2019 USENIX Annual Technical Conference (USENIX ATC 19). USENIX Association, Renton, WA, 661-674. https:\/\/www.usenix.org\/conference\/atc19\/presentation\/brokhman"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/IISWC.2012.6402918"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1145\/3620666.3651353"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1145\/3456727.3463766"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/IISWC.2009.5306797"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1145\/2931088.2931091"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1145\/3307650.3322224"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS47924.2020.00054"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1145\/1735971.1736059"},{"key":"e_1_3_2_1_17_1","volume-title":"relax","author":"NVM Express Work Group","unstructured":"NVM Express Work Group. relax. NVM Express Specifications. https:\/\/nvmexpress.org\/specifications\/"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/InPar.2012.6339596"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1145\/3205289.3205291"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1145\/3373376.3378529"},{"key":"e_1_3_2_1_21_1","unstructured":"Linux. 2003. POSIX asynchronous I\/O overview. https:\/\/man7.org\/linux\/man-pages\/man7\/aio.7.html"},{"key":"e_1_3_2_1_22_1","unstructured":"Linux. relax. Linux Heterogeneous Memory Management. https:\/\/www.kernel.org\/doc\/html\/latest\/mm\/hmm.html"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1109\/SC.2018.00035"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1145\/3462545"},{"key":"e_1_3_2_1_25_1","volume-title":"MPI: A Message-Passing Interface Standard Version 5.0. https:\/\/www.mpi-forum.org\/docs\/mpi-5.0\/mpi50-report.pdf","author":"Interface Forum Message Passing","year":"2025","unstructured":"Message Passing Interface Forum. 2025. MPI: A Message-Passing Interface Standard Version 5.0. https:\/\/www.mpi-forum.org\/docs\/mpi-5.0\/mpi50-report.pdf"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1145\/248052.248106"},{"key":"e_1_3_2_1_27_1","unstructured":"Micron. 2023. Micron 7450 NVMe SSD. https:\/\/www.micron.com\/products\/storage\/ssd\/data-center-ssd\/7450-ssd"},{"key":"e_1_3_2_1_28_1","unstructured":"Microsoft. 2022. DirectX DirectStorage. https:\/\/devblogs.microsoft.com\/directx\/directstorage-api-downloads\/"},{"key":"e_1_3_2_1_29_1","volume-title":"Zaid Qureshi, Jinjun Xiong, Eiman Ebrahimi, and Wen mei Hwu.","author":"Min Seung Won","year":"2021","unstructured":"Seung Won Min, Vikram Sharma Mailthody, Zaid Qureshi, Jinjun Xiong, Eiman Ebrahimi, and Wen mei Hwu. 2021. EMOGI: Efficient Memory-access for Out-of-memory Graph-traversal In GPUs. arXiv:2006.06890 [cs.DC] https:\/\/arxiv.org\/abs\/2006.06890"},{"key":"e_1_3_2_1_30_1","unstructured":"Nvidia. 2015. GPU Pro Tip: CUDA 7 Streams Simplify Concurrency. https:\/\/developer.nvidia.com\/blog\/gpu-pro-tip-cuda-7-streams-simplify-concurrency\/"},{"key":"e_1_3_2_1_31_1","unstructured":"Nvidia. 2017. Unified Memory for CUDA Beginners. https:\/\/developer.nvidia.com\/blog\/unified-memory-cuda-beginners\/"},{"key":"e_1_3_2_1_32_1","unstructured":"Nvidia. 2019. GPUDirect Storage: A Direct Path Between Storage and GPU Memory. https:\/\/developer.nvidia.com\/blog\/gpudirect-storage\/"},{"key":"e_1_3_2_1_33_1","unstructured":"Nvidia. 2020. Controlling Data Movement to Boost Performance on the NVIDIA Ampere Architecture. https:\/\/developer.nvidia.com\/blog\/controlling-data-movement-to-boost-performance-on-ampere-architecture\/"},{"key":"e_1_3_2_1_34_1","unstructured":"Nvidia. 2022. NVIDIA Hopper Architecture In-Depth. https:\/\/developer.nvidia.com\/blog\/nvidia-hopper-architecture-in-depth\/"},{"key":"e_1_3_2_1_35_1","unstructured":"Nvidia. 2024. Green Contexts. https:\/\/docs.nvidia.com\/cuda\/cuda-driver-api\/group__CUDA__GREEN__CONTEXTS.html"},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1145\/1778765.1778803"},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1145\/3575693.3575748"},{"key":"e_1_3_2_1_38_1","unstructured":"Samsung. 2022. Samsung 990 Pro. https:\/\/semiconductor.samsung.com\/consumer-storage\/internal-ssd\/990-pro\/"},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1145\/3007787.3001200"},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1145\/2451116.2451169"},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1145\/2366145.2366180"},{"key":"e_1_3_2_1_42_1","unstructured":"Sysbench. 2004. . https:\/\/github.com\/akopytov\/sysbench"},{"key":"e_1_3_2_1_43_1","volume-title":"Gullfoss: Accelerating and simplifying data movement among heterogeneous computing and storage resources. UC San Diego: Department of Computer Science & Engineering","author":"Tseng Hung-Wei","year":"2015","unstructured":"Hung-Wei Tseng, Yang Liu, Mark Gahagan, Jing Li, Yanqin Jing, and Steven Swanson. 2015. Gullfoss: Accelerating and simplifying data movement among heterogeneous computing and storage resources. UC San Diego: Department of Computer Science & Engineering (2015). https:\/\/escholarship.org\/uc\/item\/324465dv"},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA.2018.00075"},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1145\/3582514.3582517"},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS.2010.5470477"},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.1145\/3297858.3304024"},{"key":"e_1_3_2_1_48_1","volume-title":"AGILE: Lightweight and Efficient Asynchronous GPU-SSD Integration. arXiv:2504.19365 [cs.DC] https:\/\/arxiv.org\/abs\/2504.19365","author":"Yang Zhuoping","year":"2025","unstructured":"Zhuoping Yang, Jinming Zhuang, Xingzhen Chen, Alex K. Jones, and Peipei Zhou. 2025. AGILE: Lightweight and Efficient Asynchronous GPU-SSD Integration. arXiv:2504.19365 [cs.DC] https:\/\/arxiv.org\/abs\/2504.19365"},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.1145\/3613424.3614309"},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"publisher","DOI":"10.1109\/PACT.2015.43"},{"key":"e_1_3_2_1_51_1","doi-asserted-by":"publisher","DOI":"10.1145\/3316781.3317827"},{"key":"e_1_3_2_1_52_1","doi-asserted-by":"publisher","DOI":"10.1145\/3577193.3593705"}],"event":{"name":"ASPLOS '26: 31st ACM International Conference on Architectural Support for Programming Languages and Operating Systems","location":"Pittsburgh PA USA","sponsor":["SIGOPS ACM Special Interest Group on Operating Systems","SIGPLAN ACM Special Interest Group on Programming Languages","SIGARCH ACM Special Interest Group on Computer Architecture","SIGBED ACM Special Interest Group on Embedded Systems"]},"container-title":["Proceedings of the 31st ACM International Conference on Architectural Support for Programming Languages and Operating Systems, Volume 2"],"original-title":[],"deposited":{"date-parts":[[2026,3,15]],"date-time":"2026-03-15T13:56:27Z","timestamp":1773582987000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3779212.3790130"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,3,22]]},"references-count":52,"alternative-id":["10.1145\/3779212.3790130","10.1145\/3779212"],"URL":"https:\/\/doi.org\/10.1145\/3779212.3790130","relation":{},"subject":[],"published":{"date-parts":[[2026,3,22]]},"assertion":[{"value":"2026-03-22","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}