{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,5]],"date-time":"2026-07-05T09:15:53Z","timestamp":1783242953743,"version":"3.54.6"},"publisher-location":"New York, NY, USA","reference-count":55,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,7,5]],"date-time":"2026-07-05T00:00:00Z","timestamp":1783209600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,7,5]]},"DOI":"10.1145\/3805760.3814919","type":"proceedings-article","created":{"date-parts":[[2026,7,5]],"date-time":"2026-07-05T08:46:13Z","timestamp":1783241173000},"page":"288-298","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["How Robustly Do LLMs Understand Execution Semantics?"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0000-5932-7022","authenticated-orcid":false,"given":"Claudio","family":"Spiess","sequence":"first","affiliation":[{"name":"University of California at Davis, Davis, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4346-5276","authenticated-orcid":false,"given":"Prem","family":"Devanbu","sequence":"additional","affiliation":[{"name":"University of California at Davis, Davis, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0771-7891","authenticated-orcid":false,"given":"Earl T.","family":"Barr","sequence":"additional","affiliation":[{"name":"University College London, London, United Kingdom"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,7,5]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.2504.01943"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.2502"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1145\/3766552"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.2206.12"},{"key":"e_1_3_2_1_5_1","unstructured":"David Bieber Rishab Goel Daniel Zheng Hugo Larochelle and Daniel Tarlow. 2022. Static prediction of runtime errors by learning to execute programs with external resource descriptions. (2022). https:\/\/arxiv.org\/abs\/2203.03771 arXiv: 2203.03771 [cs.LG]."},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.2509.21173"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1145\/3643742"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.2506"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1145\/3540250.3549162"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICSE55347.2025.00012"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1145\/3643780"},{"key":"e_1_3_2_1_12_1","first-page":"8890","volume-title":"Proceedings of the 42nd International Conference on Machine Learning. International Conference on Machine Learning. PMLR, (Oct. 6, 2025","author":"Chen Simin","year":"2025","unstructured":"Simin Chen, Pranav Pusarla, and Baishakhi Ray. 2025. DyCodeEval: Dynamic Benchmarking of Reasoning Capabilities in Code Large Language Models Under Data Contamination. In Proceedings of the 42nd International Conference on Machine Learning. International Conference on Machine Learning. PMLR, (Oct. 6, 2025), 8890-8909. Retrieved Jan. 9, 2026 from https:\/\/proceedings.mlr.p ress\/v267\/chen25ba.html."},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","unstructured":"DeepSeek-AI et al. 2025. DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning. Nature 645 8081 (Sept. 2025) 633-638. doi:10.1038\/s41586-025-09422-z. 10.1038\/s41586-025-09422-z","DOI":"10.1038\/s41586-025-09422-z"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICPC66645.2025.00055"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/SANER56733.2023.00043"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","unstructured":"Aaron Grattafiori et al. 2024. The Llama 3 Herd of Models. arXiv (Nov. 2024). arXiv: 2407.21783 [cs]. doi:10.48550\/arXiv.2407.21783. 10.48550\/arXiv.2407.21783","DOI":"10.48550\/arXiv.2407.21783"},{"key":"e_1_3_2_1_17_1","first-page":"16568","volume-title":"Understanding and Execution. In Proceedings of the 41st International Conference on Machine Learning. PMLR, (July","author":"Gu Alex","year":"2024","unstructured":"Alex Gu, Baptiste Roziere, Hugh James Leather, Armando Solar-Lezama, Gabriel Synnaeve, and Sida Wang. 2024. CRUXEval: A Benchmark for Code Reasoning, Understanding and Execution. In Proceedings of the 41st International Conference on Machine Learning. PMLR, (July 2024), 16568-16621."},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.5555\/3780338.3781132"},{"key":"e_1_3_2_1_19_1","volume-title":"Dynamic Benchmarking for Code Language Models. In NeurIPS 2025 Fourth Workshop on Deep Learning for Code. (Nov. 25","author":"Guan Batu","year":"2025","unstructured":"Batu Guan, Xiao Wu, Yuanyuan Yuan, and Shaohua Li. 2025. Is Your Benchmark Still Useful? Dynamic Benchmarking for Code Language Models. In NeurIPS 2025 Fourth Workshop on Deep Learning for Code. (Nov. 25, 2025). Retrieved Mar. 17, 2026 from https:\/\/openreview.net\/forum?id=MNRLUoaUbw."},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-981-95-4969-6_18"},{"key":"e_1_3_2_1_21_1","first-page":"2194","volume-title":"Measuring Massive Multitask Language Understanding. In International Conference on Learning Representations. (Oct.","author":"Hendrycks Dan","year":"2020","unstructured":"Dan Hendrycks, Collin Burns, Steven Basart, Andy Zou, Mantas Mazeika, Dawn Song, and Jacob Steinhardt. 2020. Measuring Massive Multitask Language Understanding. In International Conference on Learning Representations. (Oct. 2020). isbn: 979-8-3313-2194-9."},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/SANER53432.2022.00070"},{"key":"e_1_3_2_1_23_1","volume-title":"Counterfactual Analysis for Code Predicates. In Proceedings of the 41st International Conference on Machine Learning. PMLR, (July 2024)","author":"Hooda Ashish","year":"2024","unstructured":"Ashish Hooda, Mihai Christodorescu, Miltiadis Allamanis, Aaron Wilson, Kassem Fawaz, and Somesh Jha. 2024. Do Large Code Models Understand Programming Concepts? Counterfactual Analysis for Code Predicates. In Proceedings of the 41st International Conference on Machine Learning. PMLR, (July 2024), 18738-18748."},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1145\/3691620.3695072"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.findings-acl.1133"},{"key":"e_1_3_2_1_26_1","first-page":"58791","volume-title":"LiveCodeBench: Holistic and Contamination Free Evaluation of Large Language Models for Code. International Conference on Representation Learning, 2025","author":"Jain Naman","year":"2025","unstructured":"Naman Jain, Alex Gu, Wen-Ding Li, Fanjia Yan, Tianjun Zhang, Sida Wang, Armando Solar-Lezama, Koushik Sen, and Ion Stoica. 2025. LiveCodeBench: Holistic and Contamination Free Evaluation of Large Language Models for Code. International Conference on Representation Learning, 2025, (May 2025), 58791-58831."},{"key":"e_1_3_2_1_27_1","first-page":"26981","volume-title":"Proceedings of the 42nd International Conference on Machine Learning. PMLR, (Oct.","author":"Javed Saqib","year":"2025","unstructured":"Saqib Javed, Hieu Le, and Mathieu Salzmann. 2025. QT-DoG: Quantization- Aware Training for Domain Generalization. In Proceedings of the 42nd International Conference on Machine Learning. PMLR, (Oct. 2025), 26981-27004."},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1145\/3747588"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1007\/BF02249043"},{"key":"e_1_3_2_1_30_1","volume-title":"The Thirty-ninth Annual Conference on Neural Information Processing Systems. (Oct.","author":"Lam Man Ho","year":"2025","unstructured":"Man Ho Lam, Chaozheng Wang, Jen-tse Huang, and Michael Lyu. 2025. Code- Crash: Exposing LLM Fragility to Misleading Natural Language in Code Reasoning. In The Thirty-ninth Annual Conference on Neural Information Processing Systems. (Oct. 2025)."},{"key":"e_1_3_2_1_31_1","first-page":"1178 1276","volume-title":"Proceedings of the 33rd ACM International Conference on the Foundations of Software Engineering. Association for Computing Machinery","author":"Liu Changshu","year":"2025","unstructured":"Changshu Liu and Reyhan Jabbarvand. 2025. A Tool for In-depth Analysis of Code Execution Reasoning of Large Language Models. In Proceedings of the 33rd ACM International Conference on the Foundations of Software Engineering. Association for Computing Machinery, New York, NY, USA, (July 2025), 1178- 1182. isbn: 979-8-4007-1276-0."},{"key":"e_1_3_2_1_32_1","first-page":"21558","volume-title":"Proceedings of the 37th International Conference on Neural Information Processing Systems (NIPS '23)","author":"Liu Jiawei","year":"2023","unstructured":"Jiawei Liu, Chunqiu Steven Xia, Yuyao Wang, and Lingming Zhang. 2023. Is Your Code Generated by ChatGPT Really Correct? Rigorous Evaluation of Large Language Models for Code Generation. In Proceedings of the 37th International Conference on Neural Information Processing Systems (NIPS '23). Curran Associates Inc., Red Hook, NY, USA, (Dec. 2023), 21558-21572."},{"key":"e_1_3_2_1_33_1","volume-title":"Neurips Safe Generative AI Workshop 2024","author":"Mu Norman","year":"2024","unstructured":"Norman Mu, Jonathan Lu, Michael Lavery, and David Wagner. 2024. A Closer Look at System Message Robustness. In Neurips Safe Generative AI Workshop 2024. (Oct. 2024)."},{"key":"e_1_3_2_1_34_1","first-page":"2614","volume-title":"CodeMMLU: A Multi-Task Benchmark for Assessing Code Understanding & Reasoning Capabilities of CodeLLMs. International Conference on Representation Learning, 2025","author":"Nguyen Dung","year":"2025","unstructured":"Dung Nguyen, Thang Phan, Nam Le Hai, Thong Doan, Nam Nguyen, Quang Pham, and Nghi Bui. 2025. CodeMMLU: A Multi-Task Benchmark for Assessing Code Understanding & Reasoning Capabilities of CodeLLMs. International Conference on Representation Learning, 2025, (May 2025), 2614-2672."},{"key":"e_1_3_2_1_35_1","volume-title":"2025 IEEE\/ACM 47th International Conference on Software Engineering (ICSE). IEEE Computer Society, 639-639","author":"Patel Smit","year":"2025","unstructured":"Smit Patel, Aashish Yadavally, Hridya Dhulipala, and Tien Nguyen. 2025. Planning a large language model for static detection of runtime errors in code snippets. In 2025 IEEE\/ACM 47th International Conference on Software Engineering (ICSE). IEEE Computer Society, 639-639."},{"key":"e_1_3_2_1_36_1","unstructured":"Julian Aron Prenner and Romain Robbes. 2025. ThrowBench: Benchmarking llms by predicting runtime exceptions. (2025). https:\/\/arxiv.org\/abs\/2503.04241 arXiv: 2503.04241 [cs.SE]."},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","unstructured":"Qwen et al. 2025. Qwen2.5 Technical Report. arXiv (Jan. 2025). arXiv: 2412.151 15 [cs]. doi:10.48550\/arXiv.2412.15115. 10.48550\/arXiv.2412.15115","DOI":"10.48550\/arXiv.2412.15115"},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.2506.00750"},{"key":"e_1_3_2_1_39_1","volume-title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models. (Apr. 27","author":"Zhihong","year":"2024","unstructured":"Zhihong Shao et al. 2024. DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models. (Apr. 27, 2024). arXiv: 2402.03300 [cs]."},{"key":"e_1_3_2_1_40_1","unstructured":"Mrinank Sharma Meg Tong Tomasz Korbak David Duvenaud Amanda Askell Samuel R Bowman Newton Cheng Esin Durmus Zac Hatfield-Dodds Scott R Johnston et al. Towards understanding sycophancy in language models. (2023). arXiv: 2310.13548."},{"key":"e_1_3_2_1_41_1","article-title":"Large language model reasoning failures","author":"Song Peiyang","year":"2026","unstructured":"Peiyang Song, Pengrui Han, and Noah Goodman. 2026. Large language model reasoning failures. Transactions on Machine Learning Research.","journal-title":"Transactions on Machine Learning Research."},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICSE55347"},{"key":"e_1_3_2_1_43_1","first-page":"0569","volume-title":"Proceedings of the IEEE\/ACM 47th International Conference on Software Engineering. IEEE Press, (Sept. 2025)","author":"Sun Weisong","year":"2025","unstructured":"Weisong Sun, Yun Miao, Yuekang Li, Hongyu Zhang, Chunrong Fang, Yi Liu, Gelei Deng, Yang Liu, and Zhenyu Chen. 2025. Source Code Summarization in the Era of Large Language Models. In Proceedings of the IEEE\/ACM 47th International Conference on Software Engineering. IEEE Press, (Sept. 2025), 1882- 1894. isbn: 979-8-3315-0569-1."},{"key":"e_1_3_2_1_44_1","volume-title":"Advances in Neural Information Processing Systems.","author":"Wang Alex","unstructured":"Alex Wang, Yada Pruksachatkun, Nikita Nangia, Amanpreet Singh, Julian Michael, Felix Hill, Omer Levy, and Samuel Bowman. 2019. SuperGLUE: A Stickier Benchmark for General-Purpose Language Understanding Systems. In Advances in Neural Information Processing Systems. Vol. 32. Curran Associates, Inc."},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","unstructured":"Shiqi Wang et al. 2023. ReCode: Robustness Evaluation of Code Generation Models. In Proceedings of the 61st Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers). Anna Rogers Jordan Boyd- Graber and Naoaki Okazaki editors. Association for Computational Linguistics Toronto Canada (July 2023) 13818-13843. doi:10.18653\/v1\/2023.acl-long .773. 10.18653\/v1\/2023.acl-long.773","DOI":"10.18653\/v1\/2023.acl-long"},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.2511.0"},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","unstructured":"Anjiang Wei et al. 2025. EquiBench: Benchmarking Large Language Models' Reasoning about Program Semantics via Equivalence Checking. In Proceedings of the 2025 Conference on Empirical Methods in Natural Language Processing. Christos Christodoulopoulos Tanmoy Chakraborty Carolyn Rose and Violet Peng editors. Association for Computational Linguistics Suzhou China (Nov. 2025) 33856-33869. isbn: 979-8-89176-332-6. doi:10.18653\/v1\/2025.emnlp-mai n.1718. 10.18653\/v1\/2025.emnlp-main.1718","DOI":"10.18653\/v1\/2025.emnlp-mai"},{"key":"e_1_3_2_1_48_1","doi-asserted-by":"publisher","DOI":"10.1109\/TR.2"},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.1109\/TSE.2017.2734091"},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.2507.05269"},{"key":"e_1_3_2_1_51_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.acl-long.1158"},{"key":"e_1_3_2_1_52_1","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.2508.05193"},{"key":"e_1_3_2_1_53_1","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.2409.12122"},{"key":"e_1_3_2_1_54_1","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.2506.23749"},{"key":"e_1_3_2_1_55_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.infsof.2025.107699"}],"event":{"name":"AIware '26: 3rd ACM International Conference on AI-Powered Software","location":"Montreal QC Canada","acronym":"AIware '26","sponsor":["SIGSOFT ACM Special Interest Group on Software Engineering"]},"container-title":["Proceedings of the 3rd ACM International Conference on AI-Powered Software"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3805760.3814919","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,5]],"date-time":"2026-07-05T08:49:46Z","timestamp":1783241386000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3805760.3814919"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,5]]},"references-count":55,"alternative-id":["10.1145\/3805760.3814919","10.1145\/3805760"],"URL":"https:\/\/doi.org\/10.1145\/3805760.3814919","relation":{},"subject":[],"published":{"date-parts":[[2026,7,5]]},"assertion":[{"value":"2026-07-05","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}