{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,21]],"date-time":"2026-07-21T17:06:56Z","timestamp":1784653616815,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":119,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,4,12]],"date-time":"2026-04-12T00:00:00Z","timestamp":1775952000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"funder":[{"DOI":"10.13039\/100000001","name":"NSF (National Science Foundation)","doi-asserted-by":"publisher","award":["2124039, and 2008660"],"award-info":[{"award-number":["2124039, and 2008660"]}],"id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100006502","name":"Defense Sciences Office, DARPA","doi-asserted-by":"publisher","award":["N66001-22-2-4037"],"award-info":[{"award-number":["N66001-22-2-4037"]}],"id":[{"id":"10.13039\/100006502","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,4,12]]},"DOI":"10.1145\/3793655.3793727","type":"proceedings-article","created":{"date-parts":[[2026,7,21]],"date-time":"2026-07-21T16:04:46Z","timestamp":1784649886000},"page":"17-28","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Assessing, Exploiting, and Mitigating Syntactic Robustness Failures in LLM-Based Code Generation"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-4793-7859","authenticated-orcid":false,"given":"Laboni","family":"Sarker","sequence":"first","affiliation":[{"name":"Computer Science, University of California Santa Barbara, Santa Barbara, California, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-8431-6695","authenticated-orcid":false,"given":"Mara","family":"Downing","sequence":"additional","affiliation":[{"name":"Computer Science, University of California Santa Barbara, Santa Barbara, California, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-0228-0069","authenticated-orcid":false,"given":"Achintya","family":"Desai","sequence":"additional","affiliation":[{"name":"Computer Science, University of California, Santa Barbara, Santa Barbara, California, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2993-1215","authenticated-orcid":false,"given":"Tevfik","family":"Bultan","sequence":"additional","affiliation":[{"name":"Computer Science, University of California, Santa Barbara, Santa Barbara, California, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,7,21]]},"reference":[{"key":"e_1_3_3_1_2_2","unstructured":"[n. d.]. AI Leaderboards are no longer useful. https:\/\/www.aisnakeoil.com\/p\/ai-leaderboards-are-no-longer-useful#footnote-3-144156093."},{"key":"e_1_3_3_1_3_2","unstructured":"[n. d.]. gcc. https:\/\/gcc.gnu.org\/."},{"key":"e_1_3_3_1_4_2","unstructured":"[n. d.]. GPT-3.5 Turbo. https:\/\/platform.openai.com\/docs\/models\/gpt-3-5-turbo."},{"key":"e_1_3_3_1_5_2","unstructured":"[n. d.]. HackerRank. https:\/\/www.hackerrank.com\/contests\/projecteuler\/challenges."},{"key":"e_1_3_3_1_6_2","unstructured":"[n. d.]. o4-mini. https:\/\/platform.openai.com\/docs\/models\/o4-mini."},{"key":"e_1_3_3_1_7_2","unstructured":"[n. d.]. State of the art code generation benchmarks. https:\/\/paperswithcode.com\/task\/code-generation."},{"key":"e_1_3_3_1_8_2","unstructured":"2025. https:\/\/github.com\/laboni68\/LLMEvaluation.git."},{"key":"e_1_3_3_1_9_2","unstructured":"Toufique Ahmed Kunal\u00a0Suresh Pai Premkumar Devanbu and Earl\u00a0T Barr. 2023. Improving few-shot prompts with relevant static analysis products. arXiv:https:\/\/arXiv.org\/abs\/2304.06815 (2023)."},{"key":"e_1_3_3_1_10_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.eacl-srw.17"},{"key":"e_1_3_3_1_11_2","unstructured":"Microsoft\u00a0Research AI4Science and Microsoft\u00a0Azure Quantum. 2023. The Impact of Large Language Models on Scientific Discovery: a Preliminary Study using GPT-4. arxiv:https:\/\/arXiv.org\/abs\/2311.07361\u00a0[cs.CL] https:\/\/arxiv.org\/abs\/2311.07361"},{"key":"e_1_3_3_1_12_2","unstructured":"Ujjwala Anantheswaran Himanshu Gupta et\u00a0al. 2024. Investigating the Robustness of LLMs on Math Word Problems. arXiv:https:\/\/arXiv.org\/abs\/2406.15444 (2024)."},{"key":"e_1_3_3_1_13_2","doi-asserted-by":"publisher","DOI":"10.1145\/3583131.3590379"},{"key":"e_1_3_3_1_14_2","doi-asserted-by":"publisher","DOI":"10.1109\/ASE51524.2021.9678706"},{"key":"e_1_3_3_1_15_2","unstructured":"Jacob Austin Augustus Odena et\u00a0al. 2021. Program synthesis with large language models. arXiv:https:\/\/arXiv.org\/abs\/2108.07732 (2021)."},{"key":"e_1_3_3_1_16_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICSE43902.2021.00039"},{"key":"e_1_3_3_1_17_2","doi-asserted-by":"crossref","unstructured":"Ali Borji. 2023. A categorical archive of chatgpt failures. arXiv:https:\/\/arXiv.org\/abs\/2302.03494 (2023).","DOI":"10.21203\/rs.3.rs-2895792\/v1"},{"key":"e_1_3_3_1_18_2","unstructured":"Rudy Bunel P Mudigonda et\u00a0al. 2020. Branch and bound for piecewise linear neural network verification. Journal of Machine Learning Research 21 (2020)."},{"key":"e_1_3_3_1_19_2","unstructured":"Alessio Buscemi. 2023. A comparative study of code generation using chatgpt 3.5 across 10 programming languages. arXiv:https:\/\/arXiv.org\/abs\/2308.04477 (2023)."},{"key":"e_1_3_3_1_20_2","unstructured":"Emir Catir Robin Claesson and Rodothea\u00a0Myrsini Tsoupidi. 2025. Evaluating Code Generation of LLMs in Advanced Computer Science Problems. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2504.14964 (2025)."},{"key":"e_1_3_3_1_21_2","doi-asserted-by":"publisher","DOI":"10.1145\/3540250.3549162"},{"key":"e_1_3_3_1_22_2","unstructured":"Mark Chen Jerry Tworek et\u00a0al. 2021. Evaluating large language models trained on code. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2107.03374 (2021)."},{"key":"e_1_3_3_1_23_2","unstructured":"Tsong\u00a0Y Chen Shing\u00a0C Cheung and Shiu\u00a0Ming Yiu. 2020. Metamorphic testing: a new approach for generating next test cases. arXiv:https:\/\/arXiv.org\/abs\/2002.12543 (2020)."},{"key":"e_1_3_3_1_24_2","unstructured":"Karl Cobbe Vineet Kosaraju et\u00a0al. 2021. Training verifiers to solve math word problems. arXiv:https:\/\/arXiv.org\/abs\/2110.14168 (2021)."},{"key":"e_1_3_3_1_25_2","doi-asserted-by":"publisher","DOI":"10.1145\/3661167.3661221"},{"key":"e_1_3_3_1_26_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.naacl-long.161"},{"key":"e_1_3_3_1_27_2","doi-asserted-by":"publisher","DOI":"10.52202\/079017-0623"},{"key":"e_1_3_3_1_28_2","doi-asserted-by":"publisher","DOI":"10.1145\/3661167.3661173"},{"key":"e_1_3_3_1_29_2","doi-asserted-by":"publisher","DOI":"10.52202\/079017-1927"},{"key":"e_1_3_3_1_30_2","doi-asserted-by":"crossref","unstructured":"Tuan Dinh Jinman Zhao et\u00a0al. 2024. Large language models of code fail at completing code with potential bugs. Advances in Neural Information Processing Systems (2024).","DOI":"10.52202\/075280-1794"},{"key":"e_1_3_3_1_31_2","doi-asserted-by":"crossref","unstructured":"Jean-Baptiste D\u00f6derlein Nguessan\u00a0Hermann Kouadio et\u00a0al. 2025. Piloting Copilot Codex and StarCoder2: Hot temperature cold prompts or black magic? J. Syst. Softw. (2025).","DOI":"10.1016\/j.jss.2025.112562"},{"key":"e_1_3_3_1_32_2","unstructured":"Mengge Du Yuntian Chen et\u00a0al. 2024. LLM4ED: Large Language Models for Automatic Equation Discovery. arXiv:https:\/\/arXiv.org\/abs\/2405.07761 (2024)."},{"key":"e_1_3_3_1_33_2","unstructured":"Abhimanyu Dubey Abhinav Jauhri et\u00a0al. 2024. The llama 3 herd of models. arXiv:https:\/\/arXiv.org\/abs\/2407.21783 (2024)."},{"key":"e_1_3_3_1_34_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICSE-FoSE59343.2023.00008"},{"key":"e_1_3_3_1_35_2","unstructured":"Fengjuan Gao Yu Wang and Ke Wang. 2023. Discrete adversarial attack to models of code. Proceedings of the ACM on Programming Languages (PLDI)."},{"key":"e_1_3_3_1_36_2","doi-asserted-by":"publisher","DOI":"10.1109\/SP.2018.00058"},{"key":"e_1_3_3_1_37_2","doi-asserted-by":"publisher","DOI":"10.5555\/1855048"},{"key":"e_1_3_3_1_38_2","unstructured":"Yuren Hao Xiang Wan and Chengxiang Zhai. 2025. An Investigation of Robustness of LLMs in Mathematical Reasoning: Benchmarking with Mathematically-Equivalent Transformation of Advanced Mathematical Problems. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2508.08833 (2025)."},{"key":"e_1_3_3_1_39_2","unstructured":"Fusen He Juan Zhai and Minxue Pan. 2024. Beyond Code Generation: Assessing Code LLM Maturity with Postconditions. arXiv:https:\/\/arXiv.org\/abs\/2407.14118 (2024)."},{"key":"e_1_3_3_1_40_2","unstructured":"Jingxuan He and Martin Vechev. 2023. Controlling large language models to generate secure and vulnerable code. arXiv e-prints (2023)."},{"key":"e_1_3_3_1_41_2","unstructured":"Joy He-Yueya Gabriel Poesia et\u00a0al. 2023. Solving math word problems by combining language models with symbolic solvers. arXiv:https:\/\/arXiv.org\/abs\/2304.09102 (2023)."},{"key":"e_1_3_3_1_42_2","unstructured":"Paul\u00a0S. Heckbert. 1998. Fourier Transforms and the Fast Fourier Transform (FFT) Algorithm. https:\/\/api.semanticscholar.org\/CorpusID:6022157"},{"key":"e_1_3_3_1_43_2","volume-title":"Proceedings of the Neural Information Processing Systems Track on Datasets and Benchmarks 1, NeurIPS Datasets and Benchmarks","author":"Hendrycks Dan","year":"2021","unstructured":"Dan Hendrycks, Steven Basart, et\u00a0al. 2021. Measuring Coding Challenge Competence With APPS. In Proceedings of the Neural Information Processing Systems Track on Datasets and Benchmarks 1, NeurIPS Datasets and Benchmarks."},{"key":"e_1_3_3_1_44_2","unstructured":"Dan Hendrycks Collin Burns et\u00a0al. 2021. Measuring Mathematical Problem Solving With the MATH Dataset. 35th Conference on Neural Information Processing Systems (NeurIPS 2021) Track on Datasets and Benchmarks (03 2021)."},{"key":"e_1_3_3_1_45_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICSTW64639.2025.10962512"},{"key":"e_1_3_3_1_46_2","doi-asserted-by":"crossref","unstructured":"Xinyi Hou Yanjie Zhao et\u00a0al. 2024. Large Language Models for Software Engineering: A Systematic Literature Review. ACM Trans. Softw. Eng. Methodol..","DOI":"10.1145\/3695988"},{"key":"e_1_3_3_1_47_2","unstructured":"Dong Huang Qingwen Bu et\u00a0al. 2023. AgentCoder: Multi-Agent-based Code Generation with Iterative Testing and Optimisation. arXiv:https:\/\/arXiv.org\/abs\/2312.13010 (2023)."},{"key":"e_1_3_3_1_48_2","doi-asserted-by":"crossref","unstructured":"A Iserles. 1989. Numerical recipes in C\u2014the art of scientific computing by WH Press BP Flannery SA Teukolsky and WT Vetterling. Pp 735.\u00a3 27\u00b7 50. 1988. ISBN 0-521-35465-X (Cambridge University Press). The Mathematical Gazette (1989).","DOI":"10.2307\/3619708"},{"key":"e_1_3_3_1_49_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-long.269"},{"key":"e_1_3_3_1_50_2","unstructured":"Slobodan Jenko Jingxuan He Niels M\u00fcndler Mark Vero and Martin Vechev. 2024. Black-Box Adversarial Attacks on LLM-Based Code Completion. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2408.02509 (2024)."},{"key":"e_1_3_3_1_51_2","doi-asserted-by":"publisher","DOI":"10.1145\/3491102.3501870"},{"key":"e_1_3_3_1_52_2","unstructured":"Juyong Jiang Fan Wang et\u00a0al. 2024. A Survey on Large Language Models for Code Generation. arxiv:https:\/\/arXiv.org\/abs\/2406.00515\u00a0[cs.CL] https:\/\/arxiv.org\/abs\/2406.00515"},{"key":"e_1_3_3_1_53_2","unstructured":"Shuyang Jiang Yuhao Wang and Yu Wang. 2023. Selfevolve: A code evolution framework via large language models. arXiv:https:\/\/arXiv.org\/abs\/2306.02907 (2023)."},{"key":"e_1_3_3_1_54_2","unstructured":"Xue Jiang Yihong Dong et\u00a0al. 2023. Self-planning Code Generation with Large Language Models. ACM Transactions on Software Engineering and Methodology."},{"key":"e_1_3_3_1_55_2","unstructured":"Sayash Kapoor Benedikt Stroebl et\u00a0al. 2025. AI Agents That Matter. Trans. Mach. Learn. Res. (2025)."},{"key":"e_1_3_3_1_56_2","doi-asserted-by":"crossref","unstructured":"Ali Kashefi and Tapan Mukerji. 2023. ChatGPT for programming numerical methods. Journal of Machine Learning for Modeling and Computing 4 2 (2023).","DOI":"10.1615\/JMachLearnModelComput.2023048492"},{"key":"e_1_3_3_1_57_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-25540-4_26"},{"key":"e_1_3_3_1_58_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-long.367"},{"key":"e_1_3_3_1_59_2","unstructured":"Saso Koceski Natasa Koceska et\u00a0al. 2023. Can ChatGPT be used for solving ordinary differential equations. International Electronic Journal of Mathematics Education 6 (12 2023). https:\/\/eprints.ugd.edu.mk\/32896\/"},{"key":"e_1_3_3_1_60_2","unstructured":"Saso Koceski Natasa Koceska Limonka Koceva\u00a0Lazarova Marija Miteva and Biljana Zlatanovska. 2023. Using ChatGPT for numerical solution of first and second order ordinary differential equations. Presentation at the 10th Jubilee International Conference of FMNS (FMNS-2023) Blagoevgrad Bulgaria."},{"key":"e_1_3_3_1_61_2","doi-asserted-by":"crossref","unstructured":"Takeshi Kojima Shixiang\u00a0Shane Gu et\u00a0al. 2022. Large language models are zero-shot reasoners. Advances in neural information processing systems (2022).","DOI":"10.52202\/068431-1613"},{"key":"e_1_3_3_1_62_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.findings-emnlp.230"},{"key":"e_1_3_3_1_63_2","unstructured":"Jia Li Ge Li Yongmin Li and Zhi Jin. 2023. Enabling programming thinking in large language models toward code generation. arXiv:https:\/\/arXiv.org\/abs\/2305.06599 (2023)."},{"key":"e_1_3_3_1_64_2","unstructured":"Jia Li Ge Li Yongmin Li and Zhi Jin. 2023. Structured chain-of-thought prompting for code generation. ACM Transactions on Software Engineering and Methodology (2023)."},{"key":"e_1_3_3_1_65_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICSE48619.2023.00179"},{"key":"e_1_3_3_1_66_2","doi-asserted-by":"crossref","unstructured":"Jia Li Yunfei Zhao et\u00a0al. 2024. AceCoder: An Effective Prompting Technique Specialized in Code Generation. ACM Trans. Softw. Eng. Methodol. (2024).","DOI":"10.1145\/3675395"},{"key":"e_1_3_3_1_67_2","unstructured":"Jia Li Yunfei Zhao Yongmin Li Ge Li and Zhi Jin. 2023. Towards enhancing in-context learning for code generation. arXiv:https:\/\/arXiv.org\/abs\/2303.17780 (2023)."},{"key":"e_1_3_3_1_68_2","unstructured":"Qintong Li Leyang Cui Xueliang Zhao Lingpeng Kong and Wei Bi. 2024. GSM-Plus: A Comprehensive Benchmark for Evaluating the Robustness of LLMs as Mathematical Problem Solvers. arXiv:https:\/\/arXiv.org\/abs\/2402.19255 (2024)."},{"key":"e_1_3_3_1_69_2","doi-asserted-by":"crossref","unstructured":"Yujia Li David Choi et\u00a0al. 2022. Competition-level code generation with alphacode. Science 378 6624 (2022) 1092\u20131097.","DOI":"10.1126\/science.abq1158"},{"key":"e_1_3_3_1_70_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-65630-9_15"},{"key":"e_1_3_3_1_71_2","unstructured":"Chao Liu Bao Xuanlin Hongyu Zhang Neng Zhang Haibo Hu Xiaohong Zhang and Meng Yan. 2023. Improving ChatGPT Prompt for Code Generation. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2305.08360 (05 2023)."},{"key":"e_1_3_3_1_72_2","unstructured":"Jiawei Liu Chunqiu\u00a0Steven Xia et\u00a0al. 2024. Is your code generated by chatgpt really correct? rigorous evaluation of large language models for code generation. Advances in Neural Information Processing Systems (2024)."},{"key":"e_1_3_3_1_73_2","doi-asserted-by":"crossref","unstructured":"Qianjun Liu Shouling Ji Changchang Liu and Chunming Wu. 2021. A practical black-box attack on source code authorship identification classifiers. IEEE Transactions on Information Forensics and Security 16 (2021) 3620\u20133633.","DOI":"10.1109\/TIFS.2021.3080507"},{"key":"e_1_3_3_1_74_2","volume-title":"International conference on data intelligence and cognitive informatics","author":"Marvin Ggaliwango","year":"2023","unstructured":"Ggaliwango Marvin, Nakayiza Hellen, Daudi Jjingo, and Joyce Nakatumba-Nabende. 2023. Prompt engineering in large language models. In International conference on data intelligence and cognitive informatics. Springer."},{"key":"e_1_3_3_1_75_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICSE48619.2023.00181"},{"key":"e_1_3_3_1_76_2","unstructured":"Fangwen Mu Lin Shi et\u00a0al. 2024. ClarifyGPT: A Framework for Enhancing LLM-Based Code Generation via Requirements Clarification. Proceedings of the ACM on Software EngineeringFSE (2024)."},{"key":"e_1_3_3_1_77_2","volume-title":"The Eleventh International Conference on Learning Representations, ICLR","author":"Nijkamp Erik","year":"2023","unstructured":"Erik Nijkamp, Bo Pang, et\u00a0al. 2023. CodeGen: An Open Large Language Model for Code with Multi-Turn Program Synthesis. In The Eleventh International Conference on Learning Representations, ICLR."},{"key":"e_1_3_3_1_78_2","unstructured":"OpenAI Josh Achiam Steven Adler et\u00a0al. 2024. GPT-4 Technical Report."},{"key":"e_1_3_3_1_79_2","doi-asserted-by":"crossref","unstructured":"Giuseppe Orlando. 2023. Assessing ChatGPT for coding finite element methods. Journal of Machine Learning for Modeling and Computing 4 (07 2023).","DOI":"10.1615\/JMachLearnModelComput.2023049326"},{"key":"e_1_3_3_1_80_2","doi-asserted-by":"crossref","unstructured":"Shuyin Ouyang Jie Zhang et\u00a0al. 2025. An Empirical Study of the Non-Determinism of ChatGPT in Code Generation. ACM Trans. Softw. Eng. Methodol..","DOI":"10.1145\/3697010"},{"key":"e_1_3_3_1_81_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.naacl-main.168"},{"key":"e_1_3_3_1_82_2","volume-title":"Forty-first International Conference on Machine Learning","author":"Pei Kexin","year":"2024","unstructured":"Kexin Pei, Weichen Li, Qirui Jin, Shuyang Liu, Scott Geng, Lorenzo Cavallaro, Junfeng Yang, and Suman Jana. 2024. Exploiting Code Symmetries for Learning Program Semantics. In Forty-first International Conference on Machine Learning."},{"key":"e_1_3_3_1_83_2","doi-asserted-by":"publisher","DOI":"10.1109\/CAIN66642.2025.00052"},{"key":"e_1_3_3_1_84_2","doi-asserted-by":"crossref","unstructured":"Yubin Qu Song Huang and Yongming Yao. 2024. A survey on robustness attacks for deep code models. Automated Software Engineering 31 (2024) 65.","DOI":"10.1007\/s10515-024-00464-7"},{"key":"e_1_3_3_1_85_2","unstructured":"Fazle Rabbi Zishuo Ding and Jinqiu Yang. 2025. A Multi-Language Perspective on the Robustness of LLM Code Generation. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2504.19108."},{"key":"e_1_3_3_1_86_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D15-1202"},{"key":"e_1_3_3_1_87_2","unstructured":"Baptiste Roziere Jonas Gehring et\u00a0al. 2023. Code llama: Open foundation models for code. arXiv:https:\/\/arXiv.org\/abs\/2308.12950 (2023)."},{"key":"e_1_3_3_1_88_2","volume-title":"International Conference on Machine Learning","author":"Shao Zhihong","year":"2023","unstructured":"Zhihong Shao, Yeyun Gong, et\u00a0al. 2023. Synthetic prompting: Generating chain-of-thought demonstrations for large language models. In International Conference on Machine Learning. PMLR."},{"key":"e_1_3_3_1_89_2","volume-title":"The Thirteenth International Conference on Learning Representations, ICLR 2025","author":"Shojaee Parshin","year":"2025","unstructured":"Parshin Shojaee, Kazem Meidani, et\u00a0al. 2025. LLM-SR: Scientific Equation Discovery via Programming with Large Language Models. In The Thirteenth International Conference on Learning Representations, ICLR 2025."},{"key":"e_1_3_3_1_90_2","doi-asserted-by":"crossref","unstructured":"Gagandeep Singh Timon Gehr Markus P\u00fcschel and Martin Vechev. 2019. An abstract domain for certifying neural networks. Proceedings of the ACM on Programming Languages 3 POPL (2019) 1\u201330.","DOI":"10.1145\/3290354"},{"key":"e_1_3_3_1_91_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICSE55347.2025.00040"},{"key":"e_1_3_3_1_92_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-65112-0_7"},{"key":"e_1_3_3_1_93_2","doi-asserted-by":"publisher","DOI":"10.1145\/3238147.3238172"},{"key":"e_1_3_3_1_94_2","doi-asserted-by":"publisher","DOI":"10.1109\/ASE56229.2023.00149"},{"key":"e_1_3_3_1_95_2","unstructured":"Jonathan Uesato Nate Kushman et\u00a0al. 2022. Solving math word problems with process-and outcome-based feedback. arXiv:https:\/\/arXiv.org\/abs\/2211.14275 (2022)."},{"key":"e_1_3_3_1_96_2","doi-asserted-by":"crossref","unstructured":"Sarthak Vishnu Sahil and Naman Garg. 2025. Unveiling the Role of GPT-4 in Solving LeetCode Programming Problems. Computer Applications in Engineering Education 33 1 (2025) e22815.","DOI":"10.1002\/cae.22815"},{"key":"e_1_3_3_1_97_2","doi-asserted-by":"publisher","DOI":"10.3389\/feduc.2023.1330486"},{"key":"e_1_3_3_1_98_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.acl-long.773"},{"key":"e_1_3_3_1_99_2","doi-asserted-by":"publisher","DOI":"10.5555\/3277203.3277323"},{"key":"e_1_3_3_1_100_2","unstructured":"Zhilong Wang Lan Zhang et\u00a0al. 2024. How Does Naming Affect LLMs on Code Analysis Tasks? arxiv:https:\/\/arXiv.org\/abs\/2307.12488\u00a0[cs.CR] https:\/\/arxiv.org\/abs\/2307.12488"},{"key":"e_1_3_3_1_101_2","doi-asserted-by":"crossref","unstructured":"Jason Wei Xuezhi Wang Dale Schuurmans Maarten Bosma Xia et\u00a0al. 2022. Chain-of-thought prompting elicits reasoning in large language models. Advances in neural information processing systems 35 (2022).","DOI":"10.52202\/068431-1800"},{"key":"e_1_3_3_1_102_2","doi-asserted-by":"crossref","unstructured":"Jules White Sam Hays Quchen Fu Jesse Spencer-Smith and Douglas\u00a0C Schmidt. 2023. Chatgpt prompt patterns for improving code quality refactoring requirements elicitation and software design. arXiv:https:\/\/arXiv.org\/abs\/2303.07839 (2023).","DOI":"10.1007\/978-3-031-55642-5_4"},{"key":"e_1_3_3_1_103_2","volume-title":"ICLR 2024 Workshop on Large Language Model (LLM) Agents","author":"Wu Yiran","year":"2024","unstructured":"Yiran Wu, Feiran Jia, et\u00a0al. 2024. MathChat: Converse to Tackle Challenging Math Problems with LLM Agents. In ICLR 2024 Workshop on Large Language Model (LLM) Agents."},{"key":"e_1_3_3_1_104_2","doi-asserted-by":"publisher","DOI":"10.1145\/3293882.3330579"},{"key":"e_1_3_3_1_105_2","doi-asserted-by":"publisher","DOI":"10.1145\/3520312.3534862"},{"key":"e_1_3_3_1_106_2","volume-title":"12th International Conference on Learning Representations","author":"Xu Xilie","year":"2024","unstructured":"Xilie Xu, Keyi Kong, et\u00a0al. 2024. An LLM can Fool Itself: A Prompt-Based Adversarial Attack. In 12th International Conference on Learning Representations."},{"key":"e_1_3_3_1_107_2","unstructured":"Ming Yan Junjie Chen et\u00a0al. 2023. Coco: Testing code generation systems via concretized instructions. arXiv:https:\/\/arXiv.org\/abs\/2308.13319 (2023)."},{"key":"e_1_3_3_1_108_2","unstructured":"Zhou Yang Zhensu Sun Terry\u00a0Yue Zhuo Prem Devanbu and David Lo. 2024. Robustness Security Privacy Explainability Efficiency and Usability of Large Language Models for Code. ArXiv abs\/2403.07506 (2024)."},{"key":"e_1_3_3_1_109_2","unstructured":"Burak Yeti\u015ftiren I\u015f\u0131k \u00d6zsoy et\u00a0al. 2023. Evaluating the Code Quality of AI-Assisted Code Generation Tools: An Empirical Study on GitHub Copilot Amazon CodeWhisperer and ChatGPT. arXiv.2304.10778 (04 2023)."},{"key":"e_1_3_3_1_110_2","unstructured":"Boning Zhang Chengxi Li and Kai Fan. 2024. MARIO Eval: Evaluate Your Math LLM with your Math LLM\u2013A mathematical dataset evaluation toolkit. arXiv:https:\/\/arXiv.org\/abs\/2404.13925 (2024)."},{"key":"e_1_3_3_1_111_2","doi-asserted-by":"crossref","unstructured":"Huangzhao Zhang Zhiyi Fu et\u00a0al. 2022. Towards Robustness of Deep Program Processing Models \u2013 Detection Estimation and Enhancement. ACM Transactions on Software Engineering and Methodology 31 (01 2022). doi:10.1145\/3511887","DOI":"10.1145\/3511887"},{"key":"e_1_3_3_1_112_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-long.737"},{"key":"e_1_3_3_1_113_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.acl-long.45"},{"key":"e_1_3_3_1_114_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.findings-emnlp.303"},{"key":"e_1_3_3_1_115_2","doi-asserted-by":"crossref","unstructured":"Li Zhong and Zilong Wang. 2024. Can LLM Replace Stack Overflow? A Study on Robustness and Reliability of Large Language Model Code Generation. Proceedings of the AAAI Conference on Artificial Intelligence (03 2024).","DOI":"10.1609\/aaai.v38i19.30185"},{"key":"e_1_3_3_1_116_2","doi-asserted-by":"crossref","unstructured":"Li Zhong Zilong Wang and Jingbo Shang. 2024. Ldb: A large language model debugger via verifying runtime execution step-by-step. arXiv:https:\/\/arXiv.org\/abs\/2402.16906 (2024).","DOI":"10.18653\/v1\/2024.findings-acl.49"},{"key":"e_1_3_3_1_117_2","doi-asserted-by":"crossref","unstructured":"Qihuang Zhong Kang Wang et\u00a0al. 2026. Achieving >97% on GSM8K: deeply understanding the problems makes LLMs better solvers for math word problems. Frontiers Comput. Sci. (2026).","DOI":"10.1007\/s11704-025-41102-z"},{"key":"e_1_3_3_1_118_2","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v38i17.29949"},{"key":"e_1_3_3_1_119_2","doi-asserted-by":"publisher","DOI":"10.1145\/3689217.3690621"},{"key":"e_1_3_3_1_120_2","unstructured":"Tomasz\u00a0G Zieli. 1992. Introduction to finite element method. (1992)."}],"event":{"name":"FORGE '26: IEEE\/ACM Third International Conference on AI Foundation Models and Software Engineering","location":"Rio de Janeiro , Brazil","acronym":"FORGE '26","sponsor":["SIGSOFT ACM Special Interest Group on Software Engineering"]},"container-title":["Proceedings of the 2026 IEEE\/ACM Third International Conference on AI Foundation Models and Software Engineering"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3793655.3793727","content-type":"application\/pdf","content-version":"vor","intended-application":"syndication"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3793655.3793727","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,21]],"date-time":"2026-07-21T16:48:05Z","timestamp":1784652485000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3793655.3793727"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,4,12]]},"references-count":119,"alternative-id":["10.1145\/3793655.3793727","10.1145\/3793655"],"URL":"https:\/\/doi.org\/10.1145\/3793655.3793727","relation":{},"subject":[],"published":{"date-parts":[[2026,4,12]]},"assertion":[{"value":"2026-07-21","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}