{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,7]],"date-time":"2026-02-07T06:39:41Z","timestamp":1770446381586,"version":"3.49.0"},"publisher-location":"New York, NY, USA","reference-count":19,"publisher":"ACM","funder":[{"name":"Austrian Research Promotion Agency (FFG)","award":["FO999903595"],"award-info":[{"award-number":["FO999903595"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,4,7]]},"DOI":"10.1145\/3731125.3731126","type":"proceedings-article","created":{"date-parts":[[2025,7,8]],"date-time":"2025-07-08T01:40:03Z","timestamp":1751938803000},"page":"1-12","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Achieving High-throughput Strided Data Movement Across GPUs"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-4028-7451","authenticated-orcid":false,"given":"Peter","family":"Thoman","sequence":"first","affiliation":[{"name":"Distributed and Parallel System Group, University of Innsbruck, Innsbruck, Austria"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7774-0344","authenticated-orcid":false,"given":"Philipp","family":"Gschwandtner","sequence":"additional","affiliation":[{"name":"Research Center High-Performance Computing, University of Innsbruck, Innsbruck, Austria"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,7,7]]},"reference":[{"key":"e_1_3_3_2_2_2","doi-asserted-by":"publisher","DOI":"10.1145\/3585341.3585351"},{"key":"e_1_3_3_2_3_2","doi-asserted-by":"publisher","DOI":"10.1145\/3388333.3388653"},{"key":"e_1_3_3_2_4_2","doi-asserted-by":"publisher","DOI":"10.1145\/2063384.2063400"},{"key":"e_1_3_3_2_5_2","doi-asserted-by":"publisher","DOI":"10.1145\/3388333.3388643"},{"key":"e_1_3_3_2_6_2","doi-asserted-by":"crossref","first-page":"375","DOI":"10.1007\/978-3-031-10461-9_26","volume-title":"Intelligent Computing","author":"Glines Mark","year":"2022","unstructured":"Mark Glines, Peter Pirgov, Lenore Mullin, and Rishi Khan. 2022. Strided DMA for\u00a0Multidimensional Array Copy and\u00a0Transpose. In Intelligent Computing, Kohei Arai (Ed.). Springer International Publishing, Cham, 375\u2013393."},{"key":"e_1_3_3_2_7_2","doi-asserted-by":"publisher","DOI":"10.1145\/1995896.1995937"},{"key":"e_1_3_3_2_8_2","doi-asserted-by":"publisher","DOI":"10.1109\/BIBM55620.2022.9995222"},{"key":"e_1_3_3_2_9_2","doi-asserted-by":"publisher","DOI":"10.1145\/3432261.3432268"},{"key":"e_1_3_3_2_10_2","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS.2014.47"},{"key":"e_1_3_3_2_11_2","doi-asserted-by":"publisher","DOI":"10.1109\/CSCI54926.2021.00336"},{"key":"e_1_3_3_2_12_2","doi-asserted-by":"publisher","DOI":"10.1145\/3624062.3624187"},{"key":"e_1_3_3_2_13_2","doi-asserted-by":"publisher","DOI":"10.1145\/3295500.3356209"},{"key":"e_1_3_3_2_14_2","doi-asserted-by":"crossref","first-page":"82","DOI":"10.1109\/CCGrid57682.2023.00018","volume-title":"2023 IEEE\/ACM 23rd International Symposium on Cluster, Cloud and Internet Computing (CCGrid)","author":"Salzmann Philip","year":"2023","unstructured":"Philip Salzmann, Fabian Knorr, Peter Thoman, Philipp Gschwandtner, Biagio Cosenza, and Thomas Fahringer. 2023. An asynchronous dataflow-driven execution model for distributed accelerator computing. In 2023 IEEE\/ACM 23rd International Symposium on Cluster, Cloud and Internet Computing (CCGrid). IEEE, 82\u201393."},{"key":"e_1_3_3_2_15_2","unstructured":"The Khronos Group. 2024. SYCL Specification Version 2020 Revision 9. https:\/\/registry.khronos.org\/SYCL\/specs\/sycl-2020\/html\/sycl-2020.html"},{"key":"e_1_3_3_2_16_2","unstructured":"The Khronos Group. 2025. SYCL Implementations. https:\/\/www.khronos.org\/sycl\/#sycl-implementations"},{"key":"e_1_3_3_2_17_2","doi-asserted-by":"publisher","unstructured":"Peter Thoman Daniel Gogl and Thomas Fahringer. 2021. Sylkan: Towards a Vulkan Compute Target Platform for SYCL(IWOCL\u201921). Association for Computing Machinery Article 3 12\u00a0pages. 10.1145\/3456669.3456683","DOI":"10.1145\/3456669.3456683"},{"key":"e_1_3_3_2_18_2","doi-asserted-by":"publisher","DOI":"10.1145\/3648115.3648136"},{"key":"e_1_3_3_2_19_2","doi-asserted-by":"publisher","DOI":"10.1109\/SC.2008.5214359"},{"key":"e_1_3_3_2_20_2","doi-asserted-by":"publisher","unstructured":"F. V\u00e1zquez J.\u00a0J. Fern\u00e1ndez and E.\u00a0M. Garz\u00f3n. 2011. A new approach for sparse matrix vector product on NVIDIA GPUs. Concurrency and Computation: Practice and Experience 23 8 (2011) 815\u2013826. 10.1002\/cpe.1658","DOI":"10.1002\/cpe.1658"}],"event":{"name":"IWOCL '25: International Workshop on OpenCL and SYCL","location":"Heidelberg Germany","acronym":"IWOCL '25"},"container-title":["Proceedings of the 13th International Workshop on OpenCL and SYCL"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3731125.3731126","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,7,8]],"date-time":"2025-07-08T01:40:13Z","timestamp":1751938813000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3731125.3731126"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,4,7]]},"references-count":19,"alternative-id":["10.1145\/3731125.3731126","10.1145\/3731125"],"URL":"https:\/\/doi.org\/10.1145\/3731125.3731126","relation":{},"subject":[],"published":{"date-parts":[[2025,4,7]]},"assertion":[{"value":"2025-07-07","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}