{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,8,23]],"date-time":"2025-08-23T00:05:04Z","timestamp":1755907504842,"version":"3.44.0"},"publisher-location":"Cham","reference-count":11,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031998560","type":"print"},{"value":"9783031998577","type":"electronic"}],"license":[{"start":{"date-parts":[[2025,8,23]],"date-time":"2025-08-23T00:00:00Z","timestamp":1755907200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,8,23]],"date-time":"2025-08-23T00:00:00Z","timestamp":1755907200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026]]},"DOI":"10.1007\/978-3-031-99857-7_18","type":"book-chapter","created":{"date-parts":[[2025,8,22]],"date-time":"2025-08-22T05:16:38Z","timestamp":1755839798000},"page":"250-263","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Tutoring LLM into a Better CUDA Optimizer"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0009-0008-4470-2748","authenticated-orcid":false,"given":"Maty\u00e1\u0161","family":"Brabec","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2231-4073","authenticated-orcid":false,"given":"Ji\u0159\u00ed","family":"Klepl","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3313-1766","authenticated-orcid":false,"given":"Michal","family":"T\u00f6pfer","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0985-8949","authenticated-orcid":false,"given":"Martin","family":"Kruli\u0161","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,8,23]]},"reference":[{"key":"18_CR1","doi-asserted-by":"publisher","unstructured":"Brabec, M., Klepl, J., T\u00f6pfer, M., Kruli\u0161, M.: Artifact of the paper: tutoring LLM into a better CUDA optimizer (2025). https:\/\/doi.org\/10.5281\/zenodo.15580207","DOI":"10.5281\/zenodo.15580207"},{"key":"18_CR2","doi-asserted-by":"crossref","unstructured":"Chen, B., Mustakin, N., Hoang, A., Fuad, S., Wong, D.: VSCuda: LLM based CUDA extension for visual studio code. In: Proceedings of the SC\u201923 Workshops of The International Conference on High Performance Computing, Network, Storage, and Analysis, pp. 11\u201317 (2023)","DOI":"10.1145\/3624062.3624064"},{"key":"18_CR3","unstructured":"Conway, J., et\u00a0al.: The game of life. Scientific American, p.\u00a04 (1970)"},{"key":"18_CR4","doi-asserted-by":"crossref","unstructured":"Fujita, T., Nakano, K., Ito, Y.: Fast simulation of Conway\u2019s game of life using bitwise parallel bulk computation on a GPU. Int. J. Found. Comput. Sci. 981\u20131003 (2016)","DOI":"10.1142\/S0129054116500404"},{"key":"18_CR5","doi-asserted-by":"crossref","unstructured":"Godoy, W.F., Valero-Lara, P., Teranishi, K., Balaprakash, P., Vetter, J.S.: Large language model evaluation for high-performance computing software development. Concurr. Comput. Pract. Exp. e8269 (2024)","DOI":"10.1002\/cpe.8269"},{"key":"18_CR6","unstructured":"Jiang, J., Wang, F., Shen, J., Kim, S., Kim, S.: A survey on large language models for code generation. arXiv preprint arXiv:2406.00515 (2024)"},{"key":"18_CR7","unstructured":"Kojima, T., Gu, S.S., Reid, M., Matsuo, Y., Iwasawa, Y.: Large language models are zero-shot reasoners. In: Advances in Neural Information Processing Systems, pp. 22199\u201322213 (2022)"},{"key":"18_CR8","doi-asserted-by":"crossref","unstructured":"Nichols, D., Davis, J.H., Xie, Z., Rajaram, A., Bhatele, A.: Can large language models write parallel code? In: Proceedings of the 33rd International Symposium on High-Performance Parallel and Distributed Computing, pp. 281\u2013294 (2024)","DOI":"10.1145\/3625549.3658689"},{"key":"18_CR9","unstructured":"OpenAI: Learning to reason with LLMs (2024). https:\/\/openai.com\/index\/learning-to-reason-with-llms\/"},{"key":"18_CR10","doi-asserted-by":"crossref","unstructured":"Palkowski, M., Gruzewski, M.: GPT-driven source-to-source transformation for generating compilable parallel CUDA code for Nussinov\u2019s algorithm. Electronics 488 (2024)","DOI":"10.3390\/electronics13030488"},{"key":"18_CR11","doi-asserted-by":"crossref","unstructured":"Zhang, J., Naruse, A., Li, X., Wang, Y.: Parallel top-k algorithms on GPU: a comprehensive study and new methods. In: Proceedings of the International Conference for High Performance Computing, Networking, Storage and Analysis, pp. 1\u201313 (2023)","DOI":"10.1145\/3581784.3607062"}],"container-title":["Lecture Notes in Computer Science","Euro-Par 2025: Parallel Processing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-99857-7_18","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,22]],"date-time":"2025-08-22T05:16:46Z","timestamp":1755839806000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-99857-7_18"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,8,23]]},"ISBN":["9783031998560","9783031998577"],"references-count":11,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-99857-7_18","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,8,23]]},"assertion":[{"value":"23 August 2025","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"The authors have no competing interests to declare that are relevant to the content of this article.","order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Disclosure of Interests"}},{"value":"Euro-Par","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Parallel Processing","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Dresden","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Germany","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2025","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"25 April 2025","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 April 2025","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"31","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"europar2025","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/2025.euro-par.org\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}