{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,23]],"date-time":"2026-04-23T07:54:34Z","timestamp":1776930874892,"version":"3.51.2"},"publisher-location":"New York, NY, USA","reference-count":27,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,11,16]]},"DOI":"10.1145\/3731599.3767361","type":"proceedings-article","created":{"date-parts":[[2025,11,7]],"date-time":"2025-11-07T16:20:02Z","timestamp":1762532402000},"page":"217-224","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Towards an Automated Workflow for Floating-Point Analysis of GPU Kernels"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-2330-9820","authenticated-orcid":false,"given":"Esteban Miguel","family":"Rangel","sequence":"first","affiliation":[{"name":"Argonne National Laboratory (ANL), Lemont, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0237-3823","authenticated-orcid":false,"given":"S. John","family":"Pennycook","sequence":"additional","affiliation":[{"name":"Intel Corporation, Santa Clara, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,11,15]]},"reference":[{"key":"e_1_3_3_1_2_2","volume-title":"Monte Carlo arithmetic: exploiting randomness in floating-point arithmetic","unstructured":"[n. d.]. Monte Carlo arithmetic: exploiting randomness in floating-point arithmetic."},{"key":"e_1_3_3_1_3_2","unstructured":"Aditya Agrawal Matthew Hedlund and Blake Hechtman. 2024. eXmY: A Data Type and Technique for Arbitrary Bit Precision Quantization. arxiv:https:\/\/arXiv.org\/abs\/2405.13938\u00a0[cs.LG] https:\/\/arxiv.org\/abs\/2405.13938"},{"key":"e_1_3_3_1_4_2","unstructured":"Benjamin\u00a0S Allen James Anchell Victor Anisimov Thomas Applencourt Abhishek Bagusetty Ramesh Balakrishnan Riccardo Balin Solomon Bekele Colleen Bertoni Cyrus Blackworth et\u00a0al. 2025. Aurora: Architecting Argonne\u2019s First Exascale Supercomputer for Accelerated Scientific Discovery. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2509.08207 (2025)."},{"key":"e_1_3_3_1_5_2","doi-asserted-by":"publisher","DOI":"10.1145\/3204919.3204933"},{"key":"e_1_3_3_1_6_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-29400-7_34"},{"key":"e_1_3_3_1_7_2","volume-title":"Probabilistic Rounding with Instruction Set Management","author":"Yohan Chatelain,","year":"2025","unstructured":"Chatelain, Yohan. 2025. Probabilistic Rounding with Instruction Set Management. https:\/\/github.com\/yohanchatelain\/prism"},{"key":"e_1_3_3_1_8_2","doi-asserted-by":"publisher","unstructured":"Stefano Cherubin and Giovanni Agosta. 2020. Tools for Reduced Precision Computation: A Survey. ACM Comput. Surv. 53 2 Article 33 (April 2020) 35\u00a0pages. 10.1145\/3381039","DOI":"10.1145\/3381039"},{"key":"e_1_3_3_1_9_2","volume-title":"Consistency of Floating-Point Results using the Intel Compiler or Why doesn\u2019t my application always give the same answer","author":"Corden Martyn\u00a0J","year":"2018","unstructured":"Martyn\u00a0J Corden and David Kreitzer. 2018. Consistency of Floating-Point Results using the Intel Compiler or Why doesn\u2019t my application always give the same answer. Technical Report. Intel Corporation."},{"key":"e_1_3_3_1_10_2","doi-asserted-by":"publisher","DOI":"10.1109\/ARITH.2016.31"},{"key":"e_1_3_3_1_11_2","unstructured":"Jack Dongarra John Gunnels Harun Bayraktar Azzam Haidar and Dan Ernst. 2024. Hardware Trends Impacting Floating-Point Computations In Scientific Applications. arxiv:https:\/\/arXiv.org\/abs\/2411.12090\u00a0[math.NA] https:\/\/arxiv.org\/abs\/2411.12090"},{"key":"e_1_3_3_1_12_2","doi-asserted-by":"publisher","DOI":"10.1145\/2503210.2504566"},{"key":"e_1_3_3_1_13_2","volume-title":"Intercept Layer for OpenCL Applications","author":"Corporation Intel","year":"2025","unstructured":"Intel Corporation. 2025. Intercept Layer for OpenCL Applications. https:\/\/github.com\/intel\/opencl-intercept-layer"},{"key":"e_1_3_3_1_14_2","volume-title":"oneAPI DPC++ Compiler","author":"Corporation Intel","year":"2025","unstructured":"Intel Corporation. 2025. oneAPI DPC++ Compiler. https:\/\/github.com\/intel\/llvm"},{"key":"e_1_3_3_1_15_2","volume-title":"Portable Computing Language","author":"J\u00e4\u00e4skel\u00e4inen Pekka","year":"2025","unstructured":"Pekka J\u00e4\u00e4skel\u00e4inen et\u00a0al. 2025. Portable Computing Language. https:\/\/github.com\/pocl\/pocl"},{"key":"e_1_3_3_1_16_2","doi-asserted-by":"crossref","unstructured":"Pekka J\u00e4\u00e4skel\u00e4inen Carlos\u00a0S\u00e1nchez de La\u00a0Lama Erik Schnetter Kalle Raiskila Jarmo Takala and Heikki Berg. 2015. pocl: A Performance-Portable OpenCL Implementation. International Journal of Parallel Programming 43 5 (2015) 752\u2013785.","DOI":"10.1007\/s10766-014-0320-y"},{"key":"e_1_3_3_1_17_2","unstructured":"Dhiraj Kalamkar Dheevatsa Mudigere Naveen Mellempudi Dipankar Das Kunal Banerjee Sasikanth Avancha Dharma\u00a0Teja Vooturi Nataraj Jammalamadaka Jianyu Huang Hector Yuen Jiyan Yang Jongsoo Park Alexander Heinecke Evangelos Georganas Sudarshan Srinivasan Abhisek Kundu Misha Smelyanskiy Bharat Kaul and Pradeep Dubey. 2019. A Study of BFLOAT16 for Deep Learning Training. arxiv:https:\/\/arXiv.org\/abs\/1905.12322\u00a0[cs.LG] https:\/\/arxiv.org\/abs\/1905.12322"},{"key":"e_1_3_3_1_18_2","volume-title":"LLVM\/SPIR-V Bi-Directional Translator","author":"Group Khronos","unstructured":"Khronos Group. [n. d.]. LLVM\/SPIR-V Bi-Directional Translator. https:\/\/github.com\/KhronosGroup\/SPIRV-LLVM-Translator"},{"key":"e_1_3_3_1_19_2","volume-title":"SPIR-V Specification, Version 1.6, Revision 6","author":"Group Khronos","year":"2025","unstructured":"Khronos Group. 2025. SPIR-V Specification, Version 1.6, Revision 6."},{"key":"e_1_3_3_1_20_2","volume-title":"SYCL 2020 Specification (revision 10)","author":"Group Khronos","year":"2025","unstructured":"Khronos Group. 2025. SYCL 2020 Specification (revision 10)."},{"key":"e_1_3_3_1_21_2","doi-asserted-by":"publisher","unstructured":"Andreas Kl\u00f6ckner Nicolas Pinto Yunsup Lee Bryan Catanzaro Paul Ivanov and Ahmed Fasih. 2012. PyCUDA and PyOpenCL: A Scripting-based Approach to GPU Run-time Code Generation. Parallel Comput. 38 3 (March 2012) 157\u2013174. 10.1016\/j.parco.2011.09.001","DOI":"10.1016\/j.parco.2011.09.001"},{"key":"e_1_3_3_1_22_2","doi-asserted-by":"publisher","DOI":"10.1145\/3330345.3330360"},{"key":"e_1_3_3_1_23_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-20656-7_12"},{"key":"e_1_3_3_1_24_2","doi-asserted-by":"publisher","DOI":"10.1109\/CGO.2004.1281665"},{"key":"e_1_3_3_1_25_2","doi-asserted-by":"publisher","DOI":"10.1145\/3581784.3607098"},{"key":"e_1_3_3_1_26_2","doi-asserted-by":"publisher","DOI":"10.1145\/3624062.3624187"},{"key":"e_1_3_3_1_27_2","unstructured":"Bita\u00a0Darvish Rouhani Ritchie Zhao Ankit More Mathew Hall Alireza Khodamoradi Summer Deng Dhruv Choudhary Marius Cornea Eric Dellinger Kristof Denolf Stosic Dusan Venmugil Elango Maximilian Golub Alexander Heinecke Phil James-Roxby Dharmesh Jani Gaurav Kolhe Martin Langhammer Ada Li Levi Melnick Maral Mesmakhosroshahi Andres Rodriguez Michael Schulte Rasoul Shafipour Lei Shao Michael Siu Pradeep Dubey Paulius Micikevicius Maxim Naumov Colin Verrilli Ralph Wittig Doug Burger and Eric Chung. 2023. Microscaling Data Formats for Deep Learning. arxiv:https:\/\/arXiv.org\/abs\/2310.10537\u00a0[cs.LG] https:\/\/arxiv.org\/abs\/2310.10537"},{"key":"e_1_3_3_1_28_2","doi-asserted-by":"crossref","unstructured":"Christian\u00a0R. Trott Damien Lebrun-Grandi\u00e9 Daniel Arndt Jan Ciesko Vinh Dang Nathan Ellingwood Rahulkumar Gayatri Evan Harvey Daisy\u00a0S. Hollman Dan Ibanez Nevin Liber Jonathan Madsen Jeff Miles David Poliakoff Amy Powell Sivasankaran Rajamanickam Mikael Simberg Dan Sunderland Bruno Turcksin and Jeremiah Wilke. 2022. Kokkos 3: Programming Model Extensions for the Exascale Era. IEEE Transactions on Parallel and Distributed Systems 33 4 (2022) 805\u2013817.","DOI":"10.1109\/TPDS.2021.3097283"}],"event":{"name":"SC Workshops '25: Workshops of the International Conference for High Performance Computing, Networking, Storage and Analysis","location":"St Louis MO USA","acronym":"SC Workshops '25","sponsor":["SIGHPC ACM Special Interest Group on High Performance Computing, Special Interest Group on High Performance Computing"]},"container-title":["Proceedings of the SC '25 Workshops of the International Conference for High Performance Computing, Networking, Storage and Analysis"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3731599.3767361","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,1,9]],"date-time":"2026-01-09T19:36:26Z","timestamp":1767987386000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3731599.3767361"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,11,15]]},"references-count":27,"alternative-id":["10.1145\/3731599.3767361","10.1145\/3731599"],"URL":"https:\/\/doi.org\/10.1145\/3731599.3767361","relation":{},"subject":[],"published":{"date-parts":[[2025,11,15]]},"assertion":[{"value":"2025-11-15","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}