{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,7]],"date-time":"2026-07-07T15:43:05Z","timestamp":1783438985570,"version":"3.54.6"},"publisher-location":"New York, NY, USA","reference-count":63,"publisher":"ACM","funder":[{"DOI":"10.13039\/501100000266","name":"Engineering and Physical Sciences Research Council","doi-asserted-by":"publisher","award":["EP\/X018202\/1, EP\/X037304\/1, EP\/X037525\/1"],"award-info":[{"award-number":["EP\/X018202\/1, EP\/X037304\/1, EP\/X037525\/1"]}],"id":[{"id":"10.13039\/501100000266","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,6,13]]},"DOI":"10.1145\/3735950.3735954","type":"proceedings-article","created":{"date-parts":[[2025,6,13]],"date-time":"2025-06-13T15:04:50Z","timestamp":1749827090000},"page":"27-40","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":2,"title":["SecureMind: A Framework for Benchmarking Large Language Models in Memory Bug Detection and Repair"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-0579-4295","authenticated-orcid":false,"given":"Huanting","family":"Wang","sequence":"first","affiliation":[{"name":"University of Leeds, Leeds, United Kingdom"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4137-0353","authenticated-orcid":false,"given":"Dejice","family":"Jacob","sequence":"additional","affiliation":[{"name":"University of Glasgow, Glasgow, United Kingdom"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-5576-5842","authenticated-orcid":false,"given":"David","family":"Kelly","sequence":"additional","affiliation":[{"name":"University of Glasgow, Glasgow, United Kingdom"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4639-436X","authenticated-orcid":false,"given":"Yehia","family":"Elkhatib","sequence":"additional","affiliation":[{"name":"University of Glasgow, Glasgow, United Kingdom"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9462-6802","authenticated-orcid":false,"given":"Jeremy","family":"Singer","sequence":"additional","affiliation":[{"name":"University of Glasgow, Glasgow, United Kingdom"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6157-0662","authenticated-orcid":false,"given":"Zheng","family":"Wang","sequence":"additional","affiliation":[{"name":"University of Leeds, Leeds, United Kingdom"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,6,13]]},"reference":[{"key":"e_1_3_2_2_1_1","unstructured":"[n. d.]. Common Vulnerabilities and Exposures (CVE). https:\/\/cve.mitre.org\/"},{"key":"e_1_3_2_2_2_1","unstructured":"[n. d.]. Infer a static program analyzer. https:\/\/fbinfer.com\/docs\/about-Infer"},{"key":"e_1_3_2_2_3_1","unstructured":"[n. d.]. National Vulnerability Database (NVD). https:\/\/nvd.nist.gov"},{"key":"e_1_3_2_2_4_1","unstructured":"2023. Cheat Sheet: Mastering Temperature and Top P in ChatGPT API (A Few Tips and Tricks on Controlling the Creativity\/Deterministic Output of Prompt Responses). https:\/\/community.openai.com\/t\/cheat-sheet-mastering-temperature-and-top-p-in-chatgpt-api-a-few-tips-and-tricks-on-controlling-the-creativity-deterministic-output-of-prompt-responses\/172683\/1 Accessed: 2025-03-17"},{"key":"e_1_3_2_2_5_1","unstructured":"2025. Bugzilla. https:\/\/www.bugzilla.org\/"},{"key":"e_1_3_2_2_6_1","unstructured":"2025. ChatGPT. https:\/\/chat.openai.com\/"},{"key":"e_1_3_2_2_7_1","unstructured":"2025. CodeQL. https:\/\/codeql.github.com\/"},{"key":"e_1_3_2_2_8_1","unstructured":"2025. Copilot. https:\/\/copilot.microsoft.com\/"},{"key":"e_1_3_2_2_9_1","unstructured":"2025. GraphQL. https:\/\/graphql.org\/"},{"key":"e_1_3_2_2_10_1","unstructured":"2025. ZNC IRC bouncer. https:\/\/github.com\/znc\/znc"},{"key":"e_1_3_2_2_11_1","volume-title":"Sebastian Borgeaud, Jean-Baptiste Alayrac, Jiahui Yu, Radu Soricut, Johan Schalkwyk, Andrew M Dai, Anja Hauth, and Katie Millican.","author":"Google Gemini Team","year":"2023","unstructured":"Gemini Team Google: Rohan Anil, Sebastian Borgeaud, Jean-Baptiste Alayrac, Jiahui Yu, Radu Soricut, Johan Schalkwyk, Andrew M Dai, Anja Hauth, and Katie Millican. 2023. Gemini: a family of highly capable multimodal models. arXiv preprint arXiv:2312.11805."},{"key":"e_1_3_2_2_12_1","volume-title":"SLaDe: A Portable Small Language Model Decompiler for Optimized Assembly. In IEEE\/ACM International Symposium on Code Generation and Optimization (CGO). 67\u201380","author":"Armengol-Estap\u00e9 Jordi","year":"2024","unstructured":"Jordi Armengol-Estap\u00e9, Jackson Woodruff, Chris Cummins, and Michael FP O\u2019Boyle. 2024. SLaDe: A Portable Small Language Model Decompiler for Optimized Assembly. In IEEE\/ACM International Symposium on Code Generation and Optimization (CGO). 67\u201380."},{"key":"e_1_3_2_2_13_1","volume-title":"31st USENIX Security Symposium (USENIX Security 22)","author":"Arp Daniel","year":"2022","unstructured":"Daniel Arp, Erwin Quiring, Feargus Pendlebury, Alexander Warnecke, Fabio Pierazzi, Christian Wressnegger, Lorenzo Cavallaro, and Konrad Rieck. 2022. Dos and don\u2019ts of machine learning in computer security. In 31st USENIX Security Symposium (USENIX Security 22). 3971\u20133988."},{"key":"e_1_3_2_2_14_1","volume-title":"Multi-lingual Evaluation of Code Generation Models. In The Eleventh International Conference on Learning Representations.","author":"Athiwaratkun Ben","year":"2022","unstructured":"Ben Athiwaratkun. 2022. Multi-lingual Evaluation of Code Generation Models. In The Eleventh International Conference on Learning Representations."},{"key":"e_1_3_2_2_15_1","doi-asserted-by":"publisher","DOI":"10.4230\/LIPIcs.ECOOP.2016.2"},{"key":"e_1_3_2_2_16_1","doi-asserted-by":"publisher","DOI":"10.1145\/3624744"},{"key":"e_1_3_2_2_17_1","doi-asserted-by":"publisher","DOI":"10.1049\/cit2.12028"},{"key":"e_1_3_2_2_18_1","volume-title":"A survey on evaluation of large language models. ACM transactions on intelligent systems and technology, 15, 3","author":"Chang Yupeng","year":"2024","unstructured":"Yupeng Chang, Xu Wang, Jindong Wang, Yuan Wu, Linyi Yang, Kaijie Zhu, Hao Chen, Xiaoyuan Yi, Cunxiang Wang, and Yidong Wang. 2024. A survey on evaluation of large language models. ACM transactions on intelligent systems and technology, 15, 3 (2024), 1\u201345."},{"key":"e_1_3_2_2_19_1","volume-title":"Jared Kaplan, Harri Edwards, Yuri Burda, Nicholas Joseph, and Greg Brockman.","author":"Chen Mark","year":"2021","unstructured":"Mark Chen, Jerry Tworek, Heewoo Jun, Qiming Yuan, Henrique Ponde De Oliveira Pinto, Jared Kaplan, Harri Edwards, Yuri Burda, Nicholas Joseph, and Greg Brockman. 2021. Evaluating large language models trained on code. arXiv preprint arXiv:2107.03374."},{"key":"e_1_3_2_2_20_1","volume-title":"Chromium Security: Memory Safety. https:\/\/www.chromium.org\/Home\/chromium-security\/memory-safety\/ Accessed","author":"Project Chromium","year":"2025","unstructured":"Chromium Project. 2025. Chromium Security: Memory Safety. https:\/\/www.chromium.org\/Home\/chromium-security\/memory-safety\/ Accessed: 14 March 2025"},{"key":"e_1_3_2_2_21_1","unstructured":"LangChain Contributors. [n. d.]. LangChain: Build context-aware reasoning applications. https:\/\/github.com\/langchain-ai\/langchain"},{"key":"e_1_3_2_2_22_1","doi-asserted-by":"publisher","DOI":"10.1145\/3708493.3712691"},{"key":"e_1_3_2_2_23_1","doi-asserted-by":"publisher","DOI":"10.1145\/3597926.3598067"},{"key":"e_1_3_2_2_24_1","doi-asserted-by":"crossref","unstructured":"Ning Ding Shengding Hu Weilin Zhao Yulin Chen Zhiyuan Liu Hai-Tao Zheng and Maosong Sun. 2021. OpenPrompt: An Open-source Framework for Prompt-learning. arXiv preprint arXiv:2111.01998.","DOI":"10.18653\/v1\/2022.acl-demo.10"},{"key":"e_1_3_2_2_25_1","volume-title":"Large language models of code fail at completing code with potential bugs. Advances in Neural Information Processing Systems, 36","author":"Dinh Tuan","year":"2024","unstructured":"Tuan Dinh, Jinman Zhao, Samson Tan, Renato Negrinho, Leonard Lausen, Sheng Zha, and George Karypis. 2024. Large language models of code fail at completing code with potential bugs. Advances in Neural Information Processing Systems, 36 (2024)."},{"key":"e_1_3_2_2_26_1","doi-asserted-by":"crossref","unstructured":"Guanting Dong Hongyi Yuan Keming Lu Chengpeng Li Mingfeng Xue Dayiheng Liu Wei Wang Zheng Yuan Chang Zhou and Jingren Zhou. 2024. How Abilities in Large Language Models are Affected by Supervised Fine-tuning Data Composition. In ACL (1).","DOI":"10.18653\/v1\/2024.acl-long.12"},{"key":"e_1_3_2_2_27_1","doi-asserted-by":"publisher","DOI":"10.1109\/MS.2016.147"},{"key":"e_1_3_2_2_28_1","volume-title":"The 6th Joint Meeting on European software engineering conference and the ACM SIGSOFT Symposium on the Foundations of Software Engineering: Companion Papers. 549\u2013552","author":"Evans Robert B","year":"2007","unstructured":"Robert B Evans and Alberto Savoia. 2007. Differential testing: a new approach to change detection. In The 6th Joint Meeting on European software engineering conference and the ACM SIGSOFT Symposium on the Foundations of Software Engineering: Companion Papers. 549\u2013552."},{"key":"e_1_3_2_2_29_1","unstructured":"Jingzhi Gong Vardan Voskanyan Paul Brookes Fan Wu Wei Jie Jie Xu Rafail Giavrimis Mike Basios Leslie Kanthan and Zheng Wang. 2025. Language Models for Code Optimization: Survey Challenges and Future Directions. arXiv preprint arXiv:2501.01277."},{"key":"e_1_3_2_2_30_1","volume-title":"d.]. MaPPing Your Model: Assessing the Impact of Adversarial Attacks on LLM-based Programming Assistants","author":"Heibel John","unstructured":"John Heibel and Daniel Lowd. [n. d.]. MaPPing Your Model: Assessing the Impact of Adversarial Attacks on LLM-based Programming Assistants. In Trustworthy Multi-modal Foundation Models and AI Agents (TiFA)."},{"key":"e_1_3_2_2_31_1","volume-title":"Turbulence: Systematically and automatically testing instruction-tuned large language models for code. arXiv preprint arXiv:2312.14856.","author":"Honarvar Shahin","year":"2023","unstructured":"Shahin Honarvar, Mark van der Wilk, and Alastair Donaldson. 2023. Turbulence: Systematically and automatically testing instruction-tuned large language models for code. arXiv preprint arXiv:2312.14856."},{"key":"e_1_3_2_2_32_1","doi-asserted-by":"publisher","DOI":"10.1145\/3660773"},{"key":"e_1_3_2_2_33_1","doi-asserted-by":"publisher","DOI":"10.1145\/3695988"},{"key":"e_1_3_2_2_34_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10462-024-10824-0"},{"key":"e_1_3_2_2_35_1","doi-asserted-by":"publisher","DOI":"10.1145\/3597503.3639138"},{"key":"e_1_3_2_2_36_1","doi-asserted-by":"publisher","DOI":"10.1145\/3672456"},{"key":"e_1_3_2_2_37_1","volume-title":"Competition-level code generation with alphacode. Science, 378, 6624","author":"Li Yujia","year":"2022","unstructured":"Yujia Li, David Choi, Junyoung Chung, Nate Kushman, Julian Schrittwieser, R\u00e9mi Leblond, Tom Eccles, James Keeling, Felix Gimeno, and Agustin Dal Lago. 2022. Competition-level code generation with alphacode. Science, 378, 6624 (2022), 1092\u20131097."},{"key":"e_1_3_2_2_38_1","volume-title":"ROUGE: A Package for Automatic Evaluation of Summaries. In Text Summarization Branches Out","author":"Lin Chin-Yew","year":"2004","unstructured":"Chin-Yew Lin. 2004. ROUGE: A Package for Automatic Evaluation of Summaries. In Text Summarization Branches Out. Association for Computational Linguistics, 74\u201381. https:\/\/aclanthology.org\/W04-1013\/"},{"key":"e_1_3_2_2_39_1","unstructured":"Aixin Liu Bei Feng Bing Xue Bingxuan Wang Bochao Wu Chengda Lu Chenggang Zhao Chengqi Deng Chenyu Zhang and Chong Ruan. 2024. Deepseek-v3 technical report. arXiv preprint arXiv:2412.19437."},{"key":"e_1_3_2_2_40_1","volume-title":"Yuyao Wang, and Lingming Zhang.","author":"Liu Jiawei","year":"2024","unstructured":"Jiawei Liu, Chunqiu Steven Xia, Yuyao Wang, and Lingming Zhang. 2024. Is your code generated by ChatGPT really correct? Rigorous evaluation of large language models for code generation. Advances in Neural Information Processing Systems, 36 (2024)."},{"key":"e_1_3_2_2_41_1","doi-asserted-by":"publisher","DOI":"10.1145\/3643674"},{"key":"e_1_3_2_2_42_1","doi-asserted-by":"publisher","DOI":"10.1145\/3453483.3454030"},{"key":"e_1_3_2_2_43_1","volume-title":"International Conference on Machine Learning. 26106\u201326128","author":"Ni Ansong","year":"2023","unstructured":"Ansong Ni, Srini Iyer, Dragomir Radev, Veselin Stoyanov, Wen-tau Yih, Sida Wang, and Xi Victoria Lin. 2023. Lever: Learning to verify language-to-code generation with execution. In International Conference on Machine Learning. 26106\u201326128."},{"key":"e_1_3_2_2_44_1","unstructured":"NIST. [n. d.]. Software Assurance Reference Dataset Project. https:\/\/samate.nist.gov\/SRD\/"},{"key":"e_1_3_2_2_45_1","doi-asserted-by":"publisher","DOI":"10.1145\/3697010"},{"key":"e_1_3_2_2_46_1","volume-title":"IEEE Symposium on Security and Privacy (SP). 754\u2013768","author":"Pearce Hammond","year":"2022","unstructured":"Hammond Pearce, Baleegh Ahmad, Benjamin Tan, Brendan Dolan-Gavitt, and Ramesh Karri. 2022. Asleep at the keyboard? assessing the security of github copilot\u2019s code contributions. In IEEE Symposium on Security and Privacy (SP). 754\u2013768."},{"key":"e_1_3_2_2_47_1","volume-title":"IEEE Symposium on Security and Privacy (SP). 2339\u20132356","author":"Pearce Hammond","year":"2023","unstructured":"Hammond Pearce, Benjamin Tan, Baleegh Ahmad, Ramesh Karri, and Brendan Dolan-Gavitt. 2023. Examining zero-shot vulnerability repair with large language models. In IEEE Symposium on Security and Privacy (SP). 2339\u20132356."},{"key":"e_1_3_2_2_48_1","unstructured":"Baolin Peng Chunyuan Li Pengcheng He Michel Galley and Jianfeng Gao. 2023. Instruction tuning with GPT-4. arXiv preprint arXiv:2304.03277."},{"key":"e_1_3_2_2_49_1","volume-title":"The 7th international student conference on advanced science and technology ICAST. 4, 1.","author":"Rahutomo Faisal","unstructured":"Faisal Rahutomo, Teruaki Kitasuka, and Masayoshi Aritsugi. 2012. Semantic cosine similarity. In The 7th international student conference on advanced science and technology ICAST. 4, 1."},{"key":"e_1_3_2_2_50_1","doi-asserted-by":"publisher","DOI":"10.1145\/3639476.3639764"},{"key":"e_1_3_2_2_51_1","volume-title":"Proceedings of the USENIX Conference on Annual Technical Conference (USENIX ATC\u201912)","author":"Serebryany Konstantin","year":"2012","unstructured":"Konstantin Serebryany, Derek Bruening, Alexander Potapenko, and Dmitry Vyukov. 2012. AddressSanitizer: a fast address sanity checker. In Proceedings of the USENIX Conference on Annual Technical Conference (USENIX ATC\u201912). USENIX Association, USA. 28."},{"key":"e_1_3_2_2_52_1","unstructured":"Christos Thrampoulidis. 2024. Implicit Optimization Bias of Next-token Prediction in Linear Models. Advances in Neural Information Processing Systems."},{"key":"e_1_3_2_2_53_1","unstructured":"Hugo Touvron Louis Martin Kevin Stone Peter Albert Amjad Almahairi Yasmine Babaei Nikolay Bashlykov Soumya Batra Prajjwal Bhargava and Shruti Bhosale. 2023. Llama 2: Open foundation and fine-tuned chat models. arXiv preprint arXiv:2307.09288."},{"key":"e_1_3_2_2_54_1","doi-asserted-by":"publisher","DOI":"10.1109\/SP54263.2024.00210"},{"key":"e_1_3_2_2_55_1","unstructured":"A Vaswani. 2017. Attention is all you need. Advances in Neural Information Processing Systems."},{"key":"e_1_3_2_2_56_1","volume-title":"An Observational Investigation of Reverse Engineers\u2019 Processes. In 29th USENIX Security Symposium (USENIX Security 20)","author":"Votipka Daniel","year":"2020","unstructured":"Daniel Votipka, Seth Rabin, Kristopher Micinski, Jeffrey S Foster, and Michelle L Mazurek. 2020. An Observational Investigation of Reverse Engineers\u2019 Processes. In 29th USENIX Security Symposium (USENIX Security 20). 1875\u20131892."},{"key":"e_1_3_2_2_57_1","doi-asserted-by":"publisher","DOI":"10.1145\/3597503.3639212"},{"key":"e_1_3_2_2_58_1","volume-title":"Chi, Quoc V Le, and Denny Zhou","author":"Wei Jason","year":"2022","unstructured":"Jason Wei, Xuezhi Wang, Dale Schuurmans, Maarten Bosma, Fei Xia, Ed Chi, Quoc V Le, and Denny Zhou. 2022. Chain-of-thought prompting elicits reasoning in large language models. Advances in neural information processing systems, 35 (2022), 24824\u201324837."},{"key":"e_1_3_2_2_59_1","doi-asserted-by":"publisher","DOI":"10.1002\/smr.2376"},{"key":"e_1_3_2_2_60_1","doi-asserted-by":"publisher","DOI":"10.1145\/3653718"},{"key":"e_1_3_2_2_61_1","doi-asserted-by":"publisher","DOI":"10.1145\/3597503.3623342"},{"key":"e_1_3_2_2_62_1","doi-asserted-by":"publisher","DOI":"10.1145\/3395351.3399346"},{"key":"e_1_3_2_2_63_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10664-024-10602-0"}],"event":{"name":"ISMM '25: 2025 ACM SIGPLAN International Symposium on Memory Management","location":"Seoul Republic of Korea","acronym":"ISMM '25","sponsor":["SIGPLAN ACM Special Interest Group on Programming Languages"]},"container-title":["Proceedings of the 2025 ACM SIGPLAN International Symposium on Memory Management"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3735950.3735954","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,7,16]],"date-time":"2025-07-16T07:01:52Z","timestamp":1752649312000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3735950.3735954"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,6,13]]},"references-count":63,"alternative-id":["10.1145\/3735950.3735954","10.1145\/3735950"],"URL":"https:\/\/doi.org\/10.1145\/3735950.3735954","relation":{},"subject":[],"published":{"date-parts":[[2025,6,13]]},"assertion":[{"value":"2025-06-13","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}