{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,8]],"date-time":"2026-07-08T08:13:16Z","timestamp":1783498396372,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":10,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,5,6]],"date-time":"2026-05-06T00:00:00Z","timestamp":1778025600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,5,6]]},"DOI":"10.1145\/3811257.3811269","type":"proceedings-article","created":{"date-parts":[[2026,7,8]],"date-time":"2026-07-08T06:38:57Z","timestamp":1783492737000},"page":"1-5","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Porting ThunderKittens from CUDA to SYCL for Intel GPU: Process, Challenges, and Lessons Learned"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0006-1340-8170","authenticated-orcid":false,"given":"Yehong","family":"Jiang","sequence":"first","affiliation":[{"name":"Intel Corporation, Santa Clara, CA, USA and Department of Electrical Engineering and Computer Science, Massachusetts Institute of Technology, CAMBRIDGE, Massachusetts, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-3770-2004","authenticated-orcid":false,"given":"Shen","family":"Chen","sequence":"additional","affiliation":[{"name":"Intel Corporation, Shanghai, Massachusetts, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3548-6066","authenticated-orcid":false,"given":"Fangwen","family":"Fu","sequence":"additional","affiliation":[{"name":"Intel Corporation, Santa Clara, CA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-8746-8448","authenticated-orcid":false,"given":"Vincent","family":"Lu","sequence":"additional","affiliation":[{"name":"Intel Corporation, Santa Clara, CA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4546-9497","authenticated-orcid":false,"given":"Yen-Kuang","family":"Chen","sequence":"additional","affiliation":[{"name":"Intel Corporation, Santa Clara, CA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-8204-6454","authenticated-orcid":false,"given":"Hong","family":"Wang","sequence":"additional","affiliation":[{"name":"Intel Corporation, Santa Clara, CA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6228-924X","authenticated-orcid":false,"given":"Xinmin","family":"Tian","sequence":"additional","affiliation":[{"name":"Intel Corporation, Santa Clara, CA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,7,7]]},"reference":[{"key":"e_1_3_3_1_2_2","doi-asserted-by":"publisher","unstructured":"William Hu Drew Wadsworth Sean Siddens Stanley Winata Daniel\u00a0Y. Fu Ryann Swann Muhammad Osama Christopher R\u00e9 and Simran Arora. 2025. HipKittens: Fast and Furious AMD Kernels. arxiv:https:\/\/arXiv.org\/abs\/2511.08083\u00a0[cs.LG] 10.48550\/arXiv.2511.08083","DOI":"10.48550\/arXiv.2511.08083"},{"key":"e_1_3_3_1_3_2","unstructured":"Intel Corporation. 2022. SYCLomatic: A New CUDA-to-SYCL Code Migration Tool. Software tool. https:\/\/www.intel.com\/content\/www\/us\/en\/developer\/tools\/oneapi\/dpc-compatibility-tool.html"},{"key":"e_1_3_3_1_4_2","volume-title":"Intel oneAPI Deep Neural Network Library (oneDNN) Developer Guide","author":"Corporation Intel","year":"2023","unstructured":"Intel Corporation. 2023. Intel oneAPI Deep Neural Network Library (oneDNN) Developer Guide. Intel Corporation, Santa Clara, CA, USA. https:\/\/www.intel.com\/content\/www\/us\/en\/developer\/tools\/oneapi\/onednn.html"},{"key":"e_1_3_3_1_5_2","volume-title":"Intel\u00ae Data Center GPU Max Series","author":"Corporation Intel","year":"2023","unstructured":"Intel Corporation. 2023. Intel\u00ae Data Center GPU Max Series. Product Overview. Intel Corporation, Santa Clara, CA, USA. https:\/\/www.intel.com\/content\/www\/us\/en\/products\/details\/discrete-gpus\/data-center-gpu\/max-series.html"},{"key":"e_1_3_3_1_6_2","unstructured":"Intel Corporation. 2024. SYCL Template Library for Accelerated Computing (SYCL-TLA). GitHub repository. https:\/\/github.com\/intel\/sycl-tla"},{"key":"e_1_3_3_1_7_2","volume-title":"SYCL 2020 Specification","author":"Group Khronos","year":"2021","unstructured":"Khronos Group. 2021. SYCL 2020 Specification. Standard. Khronos Group, Inc., Beaverton, OR, USA. https:\/\/www.khronos.org\/registry\/SYCL\/specs\/sycl-2020\/html\/sycl-2020.html"},{"key":"e_1_3_3_1_8_2","volume-title":"NVIDIA H100 Tensor Core GPU Architecture","author":"Corporation NVIDIA","year":"2022","unstructured":"NVIDIA Corporation. 2022. NVIDIA H100 Tensor Core GPU Architecture. Whitepaper. NVIDIA Corporation, Santa Clara, CA, USA. https:\/\/resources.nvidia.com\/en-us-hopper-architecture\/nvidia-h100-tensor-c"},{"key":"e_1_3_3_1_9_2","volume-title":"CUDA C++ Programming Guide","author":"Corporation NVIDIA","year":"2023","unstructured":"NVIDIA Corporation. 2023. CUDA C++ Programming Guide. NVIDIA Corporation, Santa Clara, CA, USA. https:\/\/docs.nvidia.com\/cuda\/cuda-c-programming-guide\/"},{"key":"e_1_3_3_1_10_2","volume-title":"Parallel Thread Execution ISA Version 8.2","author":"Corporation NVIDIA","year":"2023","unstructured":"NVIDIA Corporation. 2023. Parallel Thread Execution ISA Version 8.2. NVIDIA Corporation, Santa Clara, CA, USA. https:\/\/docs.nvidia.com\/cuda\/parallel-thread-execution\/"},{"key":"e_1_3_3_1_11_2","series-title":"(ICLR \u201925)","volume-title":"Proceedings of the 13th International Conference on Learning Representations","author":"Spector Benjamin\u00a0F.","year":"2025","unstructured":"Benjamin\u00a0F. Spector, Simran Arora, Aaryan Singhal, Arjun Parthasarathy, Daniel\u00a0Y. Fu, and Christopher R\u00e9. 2025. ThunderKittens: Simple, Fast, and Adorable AI Kernels. In Proceedings of the 13th International Conference on Learning Representations(ICLR \u201925). OpenReview.net, Singapore, 24\u00a0pages. https:\/\/openreview.net\/forum?id=0fJfVOSUra"}],"event":{"name":"IWOCL '26: International Workshop on OpenCL and SYCL","location":"Heilbronn Germany","acronym":"IWOCL '26"},"container-title":["Proceedings of the International Workshop on OpenCL and SYCL"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3811257.3811269","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,8]],"date-time":"2026-07-08T07:13:49Z","timestamp":1783494829000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3811257.3811269"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,5,6]]},"references-count":10,"alternative-id":["10.1145\/3811257.3811269","10.1145\/3811257"],"URL":"https:\/\/doi.org\/10.1145\/3811257.3811269","relation":{},"subject":[],"published":{"date-parts":[[2026,5,6]]},"assertion":[{"value":"2026-07-07","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}