{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,9]],"date-time":"2026-06-09T08:43:06Z","timestamp":1780994586846,"version":"3.54.1"},"publisher-location":"New York, NY, USA","reference-count":43,"publisher":"ACM","funder":[{"name":"Bavarian Ministry for Economic Affairs, Regional Development and Energy (StMWi)","award":["DIE-2209-0008, DIE-2209-0009"],"award-info":[{"award-number":["DIE-2209-0008, DIE-2209-0009"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,6,21]]},"DOI":"10.1145\/3695053.3731060","type":"proceedings-article","created":{"date-parts":[[2025,6,20]],"date-time":"2025-06-20T16:46:17Z","timestamp":1750437977000},"page":"1777-1791","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":2,"title":["GPUs All Grown-Up: Fully Device-Driven SpMV Using GPU Work Graphs"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0006-6874-1747","authenticated-orcid":false,"given":"Fabian","family":"Wildgrube","sequence":"first","affiliation":[{"name":"AMD, Munich, Germany"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4587-3716","authenticated-orcid":false,"given":"Pete","family":"Ehrett","sequence":"additional","affiliation":[{"name":"AMD, Austin, TX, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-9457-0424","authenticated-orcid":false,"given":"Paul","family":"Trojahn","sequence":"additional","affiliation":[{"name":"AMD, Munich, Germany"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9979-7579","authenticated-orcid":false,"given":"Richard","family":"Membarth","sequence":"additional","affiliation":[{"name":"THI, Ingolstadt, Germany and DFKI, Saarbr\u00fccken, Germany"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5444-6521","authenticated-orcid":false,"given":"Bradford","family":"Beckmann","sequence":"additional","affiliation":[{"name":"AMD, Bellevue, WA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-4346-8334","authenticated-orcid":false,"given":"Dominik","family":"Baumeister","sequence":"additional","affiliation":[{"name":"AMD, Munich, Germany"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4689-2932","authenticated-orcid":false,"given":"Matth\u00e4us","family":"Chajdas","sequence":"additional","affiliation":[{"name":"AMD (Now at Intel), Munich, Germany"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,6,20]]},"reference":[{"key":"e_1_3_3_2_2_2","volume-title":"Getting Started with CUDA Graphs","year":"2019","unstructured":"2019. Getting Started with CUDA Graphs. Retrieved 2024-11-06 from https:\/\/developer.nvidia.com\/blog\/cuda-graphs\/"},{"key":"e_1_3_3_2_3_2","volume-title":"enqueue_kernel Manual Page","year":"2023","unstructured":"2023. enqueue_kernel Manual Page. Retrieved 2024-11-06 from https:\/\/registry.khronos.org\/OpenCL\/sdk\/3.0\/docs\/man\/html\/enqueue_kernel.html"},{"key":"e_1_3_3_2_4_2","volume-title":"The OpenCL C Specification","year":"2024","unstructured":"2024. The OpenCL C Specification. Retrieved 2024-11-06 from https:\/\/registry.khronos.org\/OpenCL\/specs\/3.0-unified\/html\/OpenCL_C.html"},{"key":"e_1_3_3_2_5_2","volume-title":"Platform Abstraction Library Repository","year":"2024","unstructured":"2024. Platform Abstraction Library Repository. Retrieved 2024-11-06 from https:\/\/github.com\/GPUOpen-Drivers\/pal"},{"key":"e_1_3_3_2_6_2","volume-title":"ROCm Documentation: Graph Management","year":"2024","unstructured":"2024. ROCm Documentation: Graph Management. Retrieved 2024-11-06 from https:\/\/rocm.docs.amd.com\/projects\/HIP\/en\/latest\/doxygen\/html\/group___graph.html"},{"key":"e_1_3_3_2_7_2","volume-title":"rocSPARSE Documentation: Types","year":"2024","unstructured":"2024. rocSPARSE Documentation: Types. Retrieved 2024-11-06 from https:\/\/rocm.docs.amd.com\/projects\/rocSPARSE\/en\/latest\/types.html#rocsparse-spmv-alg"},{"key":"e_1_3_3_2_8_2","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA52012.2021.00034"},{"key":"e_1_3_3_2_9_2","volume-title":"CUDA Dynamic Parallelism API and Principles","author":"Adinets Andy","year":"2014","unstructured":"Andy Adinets. 2014. CUDA Dynamic Parallelism API and Principles. NVIDIA. Retrieved 2024-11-06 from https:\/\/developer.nvidia.com\/blog\/cuda-dynamic-parallelism-api-principles"},{"key":"e_1_3_3_2_10_2","doi-asserted-by":"publisher","unstructured":"H.\u00a0M. Aktulga J.\u00a0C. Fogarty S.\u00a0A. Pandit and A.\u00a0Y. Grama. 2012. Parallel reactive molecular dynamics: Numerical methods and algorithmic techniques. Parallel Comput. 38 4\u20135 (apr 2012) 245\u2013259. 10.1016\/j.parco.2011.08.005","DOI":"10.1016\/j.parco.2011.08.005"},{"key":"e_1_3_3_2_11_2","doi-asserted-by":"publisher","DOI":"10.1109\/SC.2014.69"},{"key":"e_1_3_3_2_12_2","doi-asserted-by":"publisher","DOI":"10.1109\/SC41406.2024.00049"},{"key":"e_1_3_3_2_13_2","unstructured":"NVIDIA Corporation. 2025. CUDA C++ Programming Guide Release: 12.8. https:\/\/docs.nvidia.com\/cuda\/pdf\/CUDA_C_Programming_Guide.pdf"},{"key":"e_1_3_3_2_14_2","doi-asserted-by":"publisher","DOI":"10.1109\/HiPC.2015.55"},{"key":"e_1_3_3_2_15_2","doi-asserted-by":"publisher","DOI":"10.1145\/3656019.3676897"},{"key":"e_1_3_3_2_16_2","doi-asserted-by":"publisher","DOI":"10.1109\/HPCC.2010.24"},{"key":"e_1_3_3_2_17_2","doi-asserted-by":"publisher","DOI":"10.1109\/PDP.2008.41"},{"key":"e_1_3_3_2_18_2","doi-asserted-by":"publisher","DOI":"10.1109\/SC.2014.68"},{"key":"e_1_3_3_2_19_2","doi-asserted-by":"publisher","DOI":"10.1109\/HPEC.2018.8547581"},{"key":"e_1_3_3_2_20_2","doi-asserted-by":"publisher","DOI":"10.5555\/540137"},{"key":"e_1_3_3_2_21_2","doi-asserted-by":"publisher","DOI":"10.1109\/PACT.2019.00033"},{"key":"e_1_3_3_2_22_2","doi-asserted-by":"publisher","unstructured":"Michael\u00a0Allen Heroux Jack Dongarra and Piotr Luszczek. 2013. HPCG Benchmark Technical Specification. (10 2013). 10.2172\/1113870","DOI":"10.2172\/1113870"},{"key":"e_1_3_3_2_23_2","doi-asserted-by":"publisher","DOI":"10.1145\/2832241.2832244"},{"key":"e_1_3_3_2_24_2","series-title":"Proceedings of Machine Learning Research","first-page":"2410","volume-title":"Proceedings of the 35th International Conference on Machine Learning","volume":"80","author":"Kalchbrenner Nal","year":"2018","unstructured":"Nal Kalchbrenner, Erich Elsen, Karen Simonyan, Seb Noury, Norman Casagrande, Edward Lockhart, Florian Stimberg, Aaron van\u00a0den Oord, Sander Dieleman, and Koray Kavukcuoglu. 2018. Efficient Neural Audio Synthesis. In Proceedings of the 35th International Conference on Machine Learning(Proceedings of Machine Learning Research, Vol.\u00a080), Jennifer Dy and Andreas Krause (Eds.). PMLR, Stockholm, Sweden, 2410\u20132419. https:\/\/proceedings.mlr.press\/v80\/kalchbrenner18a.html"},{"key":"e_1_3_3_2_25_2","unstructured":"Aditya Kashi Hao Lu Wesley Brewer David Rogers Michael Matheson Mallikarjun Shankar and Feiyi Wang. 2024. Mixed-precision numerics in scientific applications: survey and perspectives. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2412.19322 (2024). https:\/\/arxiv.org\/pdf\/2412.19322"},{"key":"e_1_3_3_2_26_2","doi-asserted-by":"publisher","unstructured":"Michael Kenzel Bernhard Kerbl Dieter Schmalstieg and Markus Steinberger. 2018. A high-performance software graphics pipeline architecture for the GPU. ACM Trans. Graph. 37 4 Article 140 (jul 2018) 15\u00a0pages. 10.1145\/3197517.3201374","DOI":"10.1145\/3197517.3201374"},{"key":"e_1_3_3_2_27_2","doi-asserted-by":"publisher","DOI":"10.1109\/HPEC.2016.7761646"},{"key":"e_1_3_3_2_28_2","unstructured":"Thomas\u00a0N Kipf and Max Welling. 2017. Semi-supervised classification with graph convolutional networks. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1609.02907 (2017)."},{"key":"e_1_3_3_2_29_2","volume-title":"The Rust Programming Language","author":"Klabnik Steve","year":"2023","unstructured":"Steve Klabnik and Carol Nichols. 2023. The Rust Programming Language. Retrieved 2024-11-06 from https:\/\/doc.rust-lang.org\/book\/index.html"},{"key":"e_1_3_3_2_30_2","doi-asserted-by":"publisher","DOI":"10.1145\/2814204.2814208"},{"key":"e_1_3_3_2_31_2","doi-asserted-by":"publisher","unstructured":"Roland Lei\u00dfa Klaas Boesche Sebastian Hack Ars\u00e8ne P\u00e9rard-Gayot Richard Membarth Philipp Slusallek Andr\u00e9 M\u00fcller and Bertil Schmidt. 2018. AnyDSL: A Partial Evaluation Framework for Programming High-Performance Libraries. Proceedings of the ACM on Programming Languages (PACMPL) 2 OOPSLA (Nov. 2018) 119:1\u2013119:30. 10.1145\/3276489 HiPEAC 2018 Paper Award.","DOI":"10.1145\/3276489"},{"key":"e_1_3_3_2_32_2","doi-asserted-by":"publisher","DOI":"10.1145\/2751205.2751209"},{"key":"e_1_3_3_2_33_2","doi-asserted-by":"publisher","DOI":"10.1109\/ASAP.2015.7245713"},{"key":"e_1_3_3_2_34_2","doi-asserted-by":"publisher","DOI":"10.22360\/SpringSim.2016.HPC.037"},{"key":"e_1_3_3_2_35_2","volume-title":"Work Graphs: Hands-On with the Future of Graphics Programming","author":"Oberberger Max","year":"2024","unstructured":"Max Oberberger. 2024. Work Graphs: Hands-On with the Future of Graphics Programming. AMD. Retrieved 2024-11-06 from https:\/\/www.youtube.com\/watch?v=ikwIpn_elgA"},{"key":"e_1_3_3_2_36_2","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA.2014.6853209"},{"key":"e_1_3_3_2_37_2","volume-title":"D3D12 Work Graphs v1.000","author":"Patel Amar","year":"2024","unstructured":"Amar Patel and Tex Riddel. 2024. D3D12 Work Graphs v1.000. Microsoft. Retrieved 2024-11-06 from https:\/\/github.com\/microsoft\/DirectX-Specs\/blob\/master\/d3d\/WorkGraphs.md"},{"key":"e_1_3_3_2_38_2","doi-asserted-by":"publisher","DOI":"10.1145\/3620665.3640410"},{"key":"e_1_3_3_2_39_2","doi-asserted-by":"publisher","unstructured":"Markus Steinberger Michael Kenzel Pedro Boechat Bernhard Kerbl Mark Dokter and Dieter Schmalstieg. 2014. Whippletree: task-based scheduling of dynamic workloads on the GPU. ACM Trans. Graph. 33 6 Article 228 (nov 2014) 11\u00a0pages. 10.1145\/2661229.2661250","DOI":"10.1145\/2661229.2661250"},{"key":"e_1_3_3_2_40_2","doi-asserted-by":"publisher","DOI":"10.1145\/3079079.3079086"},{"key":"e_1_3_3_2_41_2","doi-asserted-by":"publisher","unstructured":"Stanley Tzeng Brandon Lloyd and John\u00a0D. Owens. 2012. A GPU Task-Parallel Model with Dependency Resolution. Computer 45 8 (2012) 34\u201341. 10.1109\/MC.2012.255","DOI":"10.1109\/MC.2012.255"},{"key":"e_1_3_3_2_42_2","doi-asserted-by":"publisher","DOI":"10.1109\/GreenCom-CPSCom.2010.102"},{"key":"e_1_3_3_2_43_2","doi-asserted-by":"publisher","DOI":"10.1145\/1362622.1362674"},{"key":"e_1_3_3_2_44_2","doi-asserted-by":"publisher","unstructured":"Carl Yang Ayd\u0131n Bulu\u00e7 and John\u00a0D. Owens. 2022. GraphBLAST: A High-Performance Linear Algebra-based Graph Framework on the GPU. ACM Trans. Math. Softw. 48 1 Article 1 (feb 2022) 51\u00a0pages. 10.1145\/3466795","DOI":"10.1145\/3466795"}],"event":{"name":"ISCA '25: Proceedings of the 52nd Annual International Symposium on Computer Architecture","location":"Tokyo Japan","acronym":"SIGARCH '25","sponsor":["SIGARCH ACM Special Interest Group on Computer Architecture"]},"container-title":["Proceedings of the 52nd Annual International Symposium on Computer Architecture"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3695053.3731060","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,21]],"date-time":"2025-06-21T11:05:03Z","timestamp":1750503903000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3695053.3731060"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,6,20]]},"references-count":43,"alternative-id":["10.1145\/3695053.3731060","10.1145\/3695053"],"URL":"https:\/\/doi.org\/10.1145\/3695053.3731060","relation":{},"subject":[],"published":{"date-parts":[[2025,6,20]]},"assertion":[{"value":"2025-06-20","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}