{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,5]],"date-time":"2026-07-05T09:15:45Z","timestamp":1783242945477,"version":"3.54.6"},"publisher-location":"New York, NY, USA","reference-count":29,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,7,5]],"date-time":"2026-07-05T00:00:00Z","timestamp":1783209600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,7,5]]},"DOI":"10.1145\/3805760.3814932","type":"proceedings-article","created":{"date-parts":[[2026,7,5]],"date-time":"2026-07-05T08:46:13Z","timestamp":1783241173000},"page":"388-396","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["SecVulEval: Context-Aware Benchmarking of LLMs for Vulnerability Detection"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-7148-2244","authenticated-orcid":false,"given":"Md Basim Uddin","family":"Ahmed","sequence":"first","affiliation":[{"name":"York University, Toronto, Canada"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0484-3972","authenticated-orcid":false,"given":"Nima Shiri","family":"Harzevili","sequence":"additional","affiliation":[{"name":"York University, Toronto, Canada"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8829-3773","authenticated-orcid":false,"given":"Jiho","family":"Shin","sequence":"additional","affiliation":[{"name":"Queen's University, Kingston, Canada"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0861-8326","authenticated-orcid":false,"given":"Hung Viet","family":"Pham","sequence":"additional","affiliation":[{"name":"York University, Toronto, Canada"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0617-2877","authenticated-orcid":false,"given":"Song","family":"Wang","sequence":"additional","affiliation":[{"name":"York University, Toronto, Canada"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,7,5]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"crossref","unstructured":"Guru Bhandari Amara Naseer and Leon Moonen. 2021. CVEfixes: automated collection of vulnerabilities and their fixes from open-source software. In pROMISE. 30-39.","DOI":"10.1145\/3475960.3475985"},{"key":"e_1_3_2_1_2_1","first-page":"3280","article-title":"Deep learning based vulnerability detection: Are we there yet","volume":"48","author":"Chakraborty Saikat","year":"2021","unstructured":"Saikat Chakraborty, Rahul Krishna, Yangruibo Ding, and Baishakhi Ray. 2021. Deep learning based vulnerability detection: Are we there yet? TSE 48, 9 (2021), 3280-3296.","journal-title":"TSE"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1145\/3607199.3607242"},{"key":"e_1_3_2_1_4_1","volume-title":"CreativEval: Evaluating Creativity of LLM-Based Hardware Code Generation. arXiv preprint arXiv:2404.08806","author":"DeLorenzo Matthew","year":"2024","unstructured":"Matthew DeLorenzo, Vasudev Gohil, and Jeyavijayan Rajendran. 2024. CreativEval: Evaluating Creativity of LLM-Based Hardware Code Generation. arXiv preprint arXiv:2404.08806 (2024)."},{"key":"e_1_3_2_1_5_1","volume-title":"Vulnerability detection with code language models: How far are we? arXiv preprint arXiv:2403.18624","author":"Ding Yangruibo","year":"2024","unstructured":"Yangruibo Ding, Yanjun Fu, Omniyyah Ibrahim, Chawin Sitawarin, Xinyun Chen, Basel Alomair, David Wagner, Baishakhi Ray, and Yizheng Chen. 2024. Vulnerability detection with code language models: How far are we? arXiv preprint arXiv:2403.18624 (2024)."},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1145\/3379597.3387501"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1145\/3524842.3528452"},{"key":"e_1_3_2_1_8_1","first-page":"103508","article-title":"BinAIV: Semantic-enhanced vulnerability detection for Linux x86 binaries","volume":"135","author":"Gu Yeming","year":"2023","unstructured":"Yeming Gu, Hui Shu, and Fei Kang. 2023. BinAIV: Semantic-enhanced vulnerability detection for Linux x86 binaries. C&S 135 (2023), 103508.","journal-title":"C&S"},{"key":"e_1_3_2_1_9_1","unstructured":"Daya Guo Qihao Zhu Dejian Yang Zhenda Xie Kai Dong Wentao Zhang Guanting Chen Xiao Bi Yu Wu YK Li et al. 2024. DeepSeek-Coder: When the Large Language Model Meets Programming-The Rise of Code Intelligence. arXiv preprint arXiv:2401.14196 (2024)."},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-70879-4_14"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1145\/3576915.3623175"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1145\/3697090.3697103"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1145\/3524842.3527949"},{"key":"e_1_3_2_1_14_1","unstructured":"Binyuan Hui Jian Yang Zeyu Cui Jiaxi Yang Dayiheng Liu Lei Zhang Tianyu Liu Jiajun Zhang Bowen Yu Kai Dang et al. 2024. Qwen2. 5-Coder Technical Report. arXiv preprint arXiv:2409.12186 (2024)."},{"key":"e_1_3_2_1_15_1","volume-title":"Llm-assisted code cleaning for training accurate code generators. arXiv preprint arXiv:2311.14904","author":"Jain Naman","year":"2023","unstructured":"Naman Jain, Tianjun Zhang, Wei-Lin Chiang, Joseph E Gonzalez, Koushik Sen, and Ion Stoica. 2023. Llm-assisted code cleaning for training accurate code generators. arXiv preprint arXiv:2311.14904 (2023)."},{"key":"e_1_3_2_1_16_1","volume-title":"Diego de las Casas, Florian Bressand, Gianna Lengyel, Guillaume Lample, Lucile Saulnier, et al.","author":"Jiang Albert Q","year":"2023","unstructured":"Albert Q Jiang, Alexandre Sablayrolles, Arthur Mensch, Chris Bamford, Devendra Singh Chaplot, Diego de las Casas, Florian Bressand, Gianna Lengyel, Guillaume Lample, Lucile Saulnier, et al. 2023. Mistral 7B. arXiv preprint arXiv:2310.06825 (2023)."},{"key":"e_1_3_2_1_17_1","volume-title":"SEC-bench: Automated Benchmarking of LLM Agents on Real-World Software Security Tasks. arXiv preprint arXiv:2506.11791","author":"Lee Hwiwon","year":"2025","unstructured":"Hwiwon Lee, Ziqi Zhang, Hanxiao Lu, and Lingming Zhang. 2025. SEC-bench: Automated Benchmarking of LLM Agents on Real-World Software Security Tasks. arXiv preprint arXiv:2506.11791 (2025)."},{"key":"e_1_3_2_1_18_1","volume-title":"More agents is all you need. arXiv preprint arXiv:2402.05120","author":"Li Junyou","year":"2024","unstructured":"Junyou Li, Qin Zhang, Yangbin Yu, Qiang Fu, and Deheng Ye. 2024. More agents is all you need. arXiv preprint arXiv:2402.05120 (2024)."},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10664-022-10267-7"},{"key":"e_1_3_2_1_20_1","unstructured":"Chao Ni Liyu Shen Xiaohu Yang Yan Zhu and Shaohua Wang. 2024. MegaVul: AC\/C++ Vulnerability Dataset with Comprehensive Code Representations. In MSR."},{"key":"e_1_3_2_1_21_1","volume-title":"Top score on the wrong exam: On benchmarking in machine learning for vulnerability detection. arXiv preprint arXiv:2408.12986","author":"Risse Niklas","year":"2024","unstructured":"Niklas Risse and Marcel B\u00f6hme. 2024. Top score on the wrong exam: On benchmarking in machine learning for vulnerability detection. arXiv preprint arXiv:2408.12986 (2024)."},{"key":"e_1_3_2_1_22_1","volume-title":"Lprotector: An llm-driven vulnerability detection system. arXiv preprint arXiv:2411.06493","author":"Sheng Ze","year":"2024","unstructured":"Ze Sheng, Fenghua Wu, Xiangwu Zuo, Chao Li, Yuxin Qiao, and Lei Hang. 2024. Lprotector: An llm-driven vulnerability detection system. arXiv preprint arXiv:2411.06493 (2024)."},{"key":"e_1_3_2_1_23_1","first-page":"1","article-title":"A Systematic Literature Review on Automated Software Vulnerability Detection Using Machine","volume":"57","author":"Harzevili Nima Shiri","year":"2024","unstructured":"Nima Shiri Harzevili, Alvine Boaye Belle, Junjie Wang, Song Wang, Zhen Ming Jiang, and Nachiappan Nagappan. 2024. A Systematic Literature Review on Automated Software Vulnerability Detection Using Machine Learning. Comput. Surveys 57, 3 (2024), 1-36.","journal-title":"Learning. Comput. Surveys"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"crossref","unstructured":"Karthik Sreedhar and Lydia Chilton. 2025. Simulating Strategic Reasoning: Comparing the Ability of Single LLMs and Multi-Agent Systems to Replicate Human Behavior. (2025).","DOI":"10.24251\/HICSS.2025.100"},{"key":"e_1_3_2_1_25_1","volume-title":"Multi-Agent Collaboration Mechanisms: A Survey of LLMs. arXiv preprint arXiv:2501.06322","author":"Tran Khanh-Tung","year":"2025","unstructured":"Khanh-Tung Tran, Dung Dao, Minh-Duong Nguyen, Quoc-Viet Pham, Barry O'Sullivan, and Hoang D Nguyen. 2025. Multi-Agent Collaboration Mechanisms: A Survey of LLMs. arXiv preprint arXiv:2501.06322 (2025)."},{"key":"e_1_3_2_1_26_1","first-page":"1282","article-title":"How effective are neural networks for fixing security vulnerabilities","author":"Wu Yi","year":"2023","unstructured":"Yi Wu, Nan Jiang, Hung Viet Pham, Thibaud Lutellier, Jordan Davis, Lin Tan, Petr Babkin, and Sameena Shah. 2023. How effective are neural networks for fixing security vulnerabilities. In ISSTA. 1282-1294.","journal-title":"ISSTA."},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1145\/3731569.3764827"},{"key":"e_1_3_2_1_28_1","first-page":"47","article-title":"Large language model for vulnerability detection: Emerging results and future directions","author":"Zhou Xin","year":"2024","unstructured":"Xin Zhou, Ting Zhang, and David Lo. 2024. Large language model for vulnerability detection: Emerging results and future directions. In ICSE. 47-51.","journal-title":"ICSE."},{"key":"e_1_3_2_1_29_1","volume-title":"Devign: Effective vulnerability identification by learning comprehensive program semantics via graph neural networks. Advances in neural information processing systems 32","author":"Zhou Yaqin","year":"2019","unstructured":"Yaqin Zhou, Shangqing Liu, Jingkai Siow, Xiaoning Du, and Yang Liu. 2019. Devign: Effective vulnerability identification by learning comprehensive program semantics via graph neural networks. Advances in neural information processing systems 32 (2019)."}],"event":{"name":"AIware '26: 3rd ACM International Conference on AI-Powered Software","location":"Montreal QC Canada","acronym":"AIware '26","sponsor":["SIGSOFT ACM Special Interest Group on Software Engineering"]},"container-title":["Proceedings of the 3rd ACM International Conference on AI-Powered Software"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3805760.3814932","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,5]],"date-time":"2026-07-05T08:46:29Z","timestamp":1783241189000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3805760.3814932"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,5]]},"references-count":29,"alternative-id":["10.1145\/3805760.3814932","10.1145\/3805760"],"URL":"https:\/\/doi.org\/10.1145\/3805760.3814932","relation":{},"subject":[],"published":{"date-parts":[[2026,7,5]]},"assertion":[{"value":"2026-07-05","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}