{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,30]],"date-time":"2026-06-30T15:44:16Z","timestamp":1782834256230,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":78,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,10,27]],"date-time":"2024-10-27T00:00:00Z","timestamp":1729987200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"National Key R&D Program of China","award":["2023YFB2703703"],"award-info":[{"award-number":["2023YFB2703703"]}]},{"name":"Tencent Basic Platform Technology Rhino-Bird Focused Research Program"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,10,27]]},"DOI":"10.1145\/3691620.3695480","type":"proceedings-article","created":{"date-parts":[[2024,10,18]],"date-time":"2024-10-18T15:39:19Z","timestamp":1729265959000},"page":"995-1006","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":8,"title":["RMCBench: Benchmarking Large Language Models' Resistance to Malicious Code"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-0192-9992","authenticated-orcid":false,"given":"Jiachi","family":"Chen","sequence":"first","affiliation":[{"name":"Sun Yat-sen University, Guangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-6825-7518","authenticated-orcid":false,"given":"Qingyuan","family":"Zhong","sequence":"additional","affiliation":[{"name":"Sun Yat-sen University, Guangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7761-7269","authenticated-orcid":false,"given":"Yanlin","family":"Wang","sequence":"additional","affiliation":[{"name":"Sun Yat-sen University, Guangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-6009-8285","authenticated-orcid":false,"given":"Kaiwen","family":"Ning","sequence":"additional","affiliation":[{"name":"Sun Yat-sen University, Guangzhou, China"},{"name":"Peng Cheng Laboratory, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9542-9456","authenticated-orcid":false,"given":"Yongkun","family":"Liu","sequence":"additional","affiliation":[{"name":"Sun Yat-sen University, Guangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1662-0063","authenticated-orcid":false,"given":"Zenan","family":"Xu","sequence":"additional","affiliation":[{"name":"Tencent AI Lab, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-9496-5917","authenticated-orcid":false,"given":"Zhe","family":"Zhao","sequence":"additional","affiliation":[{"name":"Tencent AI Lab, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9165-8331","authenticated-orcid":false,"given":"Ting","family":"Chen","sequence":"additional","affiliation":[{"name":"University of Electronic Science and Technology of China, Sichuan, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7878-4330","authenticated-orcid":false,"given":"Zibin","family":"Zheng","sequence":"additional","affiliation":[{"name":"Sun Yat-sen University, Guangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2024,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"2023. DAN (Do Anything Now). https:\/\/www.reddit.com\/r\/ChatGPTPromptGenius\/comments\/106azp6\/dan_do_anything_now\/"},{"key":"e_1_3_2_1_2_1","unstructured":"2023. Email Phishing Attacks Up 1265% Since ChatGPT Launched: Slash-Next. https:\/\/decrypt.co\/203564\/since-chatgpt-launch-phishing-emails-are-up-1265-slashnext"},{"key":"e_1_3_2_1_3_1","unstructured":"2023. how to use infilling feature in starcoder. https:\/\/github.com\/bigcode-project\/starcoder\/issues\/99"},{"key":"e_1_3_2_1_4_1","unstructured":"2024. Confidence interval. (2024). https:\/\/en.wikipedia.org\/wiki\/Confidence_interval"},{"key":"e_1_3_2_1_5_1","unstructured":"2024. Denial-of-service attack. https:\/\/en.wikipedia.org\/wiki\/Denial-of-service_attack"},{"key":"e_1_3_2_1_6_1","unstructured":"2024. Github. https:\/\/github.com\/"},{"key":"e_1_3_2_1_7_1","unstructured":"2024. GPT-3.5 Turbo. https:\/\/platform.openai.com\/docs\/models\/gpt-3-5-turbo"},{"key":"e_1_3_2_1_8_1","unstructured":"2024. Malware. https:\/\/en.wikipedia.org\/wiki\/Malware"},{"key":"e_1_3_2_1_9_1","volume-title":"https:\/\/perspectiveapi.com\/","year":"2024","unstructured":"2024. PerspectiveApi. (2024). https:\/\/perspectiveapi.com\/"},{"key":"e_1_3_2_1_10_1","unstructured":"2024. Sample size calculator. (2024). https:\/\/www.surveysystem.com\/sscalc.htm"},{"key":"e_1_3_2_1_11_1","volume-title":"Diogo Almeida, Janko Altenschmidt, Sam Altman, Shyamal Anadkat, et al.","author":"Achiam Josh","year":"2023","unstructured":"Josh Achiam, Steven Adler, Sandhini Agarwal, Lama Ahmad, Ilge Akkaya, Florencia Leoni Aleman, Diogo Almeida, Janko Altenschmidt, Sam Altman, Shyamal Anadkat, et al. 2023. Gpt-4 technical report. arXiv preprint arXiv:2303.08774 (2023)."},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1145\/3551349.3559555"},{"key":"e_1_3_2_1_13_1","unstructured":"AI@Meta. 2024. Llama 3 Model Card. (2024). https:\/\/github.com\/meta-llama\/llama3\/blob\/main\/MODEL_CARD.md"},{"key":"e_1_3_2_1_14_1","unstructured":"Alex Albert. 2023. jailbreakchat. https:\/\/www.jailbreakchat.com\/"},{"key":"e_1_3_2_1_15_1","unstructured":"Amanda Askell Yuntao Bai Anna Chen Dawn Drain Deep Ganguli Tom Henighan Andy Jones Nicholas Joseph Ben Mann Nova DasSarma et al. 2021. A general language assistant as a laboratory for alignment. arXiv preprint arXiv:2112.00861 (2021)."},{"key":"e_1_3_2_1_16_1","volume-title":"Zijian Wang, Xiaopeng Li, Yuchen Tian, Ming Tan, Wasi Uddin Ahmad, Shiqi Wang, Qing Sun, Mingyue Shang, et al.","author":"Athiwaratkun Ben","year":"2022","unstructured":"Ben Athiwaratkun, Sanjay Krishna Gouda, Zijian Wang, Xiaopeng Li, Yuchen Tian, Ming Tan, Wasi Uddin Ahmad, Shiqi Wang, Qing Sun, Mingyue Shang, et al. 2022. Multi-lingual evaluation of code generation models. arXiv preprint arXiv:2210.14868 (2022)."},{"key":"e_1_3_2_1_17_1","unstructured":"Jacob Austin Augustus Odena Maxwell Nye Maarten Bosma Henryk Michalewski David Dohan Ellen Jiang Carrie Cai Michael Terry Quoc Le et al. 2021. Program synthesis with large language models. arXiv preprint arXiv:2108.07732 (2021)."},{"key":"e_1_3_2_1_18_1","unstructured":"Yuntao Bai Andy Jones Kamal Ndousse Amanda Askell Anna Chen Nova DasSarma Dawn Drain Stanislav Fort Deep Ganguli Tom Henighan et al. 2022. Training a helpful and harmless assistant with reinforcement learning from human feedback. arXiv preprint arXiv:2204.05862 (2022)."},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.58496\/MJCSC\/2023\/002"},{"key":"e_1_3_2_1_20_1","unstructured":"Tom Brown Benjamin Mann Nick Ryder Melanie Subbiah Jared D Kaplan Prafulla Dhariwal Arvind Neelakantan Pranav Shyam Girish Sastry Amanda Askell et al. 2020. Language models are few-shot learners. Advances in neural information processing systems 33 (2020) 1877--1901."},{"key":"e_1_3_2_1_21_1","volume-title":"When chatgpt meets smart contract vulnerability detection: How far are we? arXiv preprint arXiv:2309.05520","author":"Chen Chong","year":"2023","unstructured":"Chong Chen, Jianzhong Su, Jiachi Chen, Yanlin Wang, Tingting Bi, Yanli Wang, Xingwei Lin, Ting Chen, and Zibin Zheng. 2023. When chatgpt meets smart contract vulnerability detection: How far are we? arXiv preprint arXiv:2309.05520 (2023)."},{"key":"e_1_3_2_1_22_1","volume-title":"Identifying Smart Contract Security Issues in Code Snippets from Stack Overflow. arXiv preprint arXiv:2407.13271","author":"Chen Jiachi","year":"2024","unstructured":"Jiachi Chen, Chong Chen, Jiang Hu, John Grundy, Yanlin Wang, Ting Chen, and Zibin Zheng. 2024. Identifying Smart Contract Security Issues in Code Snippets from Stack Overflow. arXiv preprint arXiv:2407.13271 (2024)."},{"key":"e_1_3_2_1_23_1","volume-title":"Jared Kaplan, Harri Edwards, Yuri Burda, Nicholas Joseph, Greg Brockman, et al.","author":"Chen Mark","year":"2021","unstructured":"Mark Chen, Jerry Tworek, Heewoo Jun, Qiming Yuan, Henrique Ponde de Oliveira Pinto, Jared Kaplan, Harri Edwards, Yuri Burda, Nicholas Joseph, Greg Brockman, et al. 2021. Evaluating large language models trained on code. arXiv preprint arXiv:2107.03374 (2021)."},{"key":"e_1_3_2_1_24_1","unstructured":"Tianyu Cui Yanling Wang Chuanpu Fu Yong Xiao Sijia Li Xinhao Deng Yunpeng Liu Qinglin Zhang Ziyi Qiu Peiyang Li et al. 2024. Risk taxonomy mitigation and assessment benchmarks of large language model systems. arXiv preprint arXiv:2401.05778 (2024)."},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.jss.2023.111734"},{"key":"e_1_3_2_1_26_1","volume-title":"Qihao Zhu.","author":"Zhenda Xie Kai Dong Dejian Yang","year":"2024","unstructured":"Dejian Yang Zhenda Xie Kai Dong Wentao Zhang Guanting Chen Xiao Bi Y. Wu Y.K. Li Fuli Luo Yingfei Xiong Wenfeng Liang Daya Guo, Qihao Zhu. 2024. DeepSeek-Coder: When the Large Language Model Meets Programming - The Rise of Code Intelligence. https:\/\/arxiv.org\/abs\/2401.14196"},{"key":"e_1_3_2_1_27_1","volume-title":"Jailbreaker: Automated jailbreak across multiple large language model chatbots. arXiv preprint arXiv:2307.08715","author":"Deng Gelei","year":"2023","unstructured":"Gelei Deng, Yi Liu, Yuekang Li, Kailong Wang, Ying Zhang, Zefeng Li, Haoyu Wang, Tianwei Zhang, and Yang Liu. 2023. Jailbreaker: Automated jailbreak across multiple large language model chatbots. arXiv preprint arXiv:2307.08715 (2023)."},{"key":"e_1_3_2_1_28_1","volume-title":"Ramesh Nallapati, Parminder Bhatia, Dan Roth, et al.","author":"Ding Yangruibo","year":"2024","unstructured":"Yangruibo Ding, Zijian Wang, Wasi Ahmad, Hantian Ding, Ming Tan, Nihal Jain, Murali Krishna Ramanathan, Ramesh Nallapati, Parminder Bhatia, Dan Roth, et al. 2024. Crosscodeeval: A diverse and multilingual benchmark for cross-file code completion. Advances in Neural Information Processing Systems 36 (2024)."},{"key":"e_1_3_2_1_29_1","volume-title":"Classeval: A manually-crafted benchmark for evaluating llms on class-level code generation. arXiv preprint arXiv:2308.01861","author":"Du Xueying","year":"2023","unstructured":"Xueying Du, Mingwei Liu, Kaixin Wang, Hanlin Wang, Junwei Liu, Yixuan Chen, Jiayi Feng, Chaofeng Sha, Xin Peng, and Yiling Lou. 2023. Classeval: A manually-crafted benchmark for evaluating llms on class-level code generation. arXiv preprint arXiv:2308.01861 (2023)."},{"key":"e_1_3_2_1_30_1","unstructured":"GeshengSunDUT S A G A R EgoAlpha YFCao. 2024. prompt-in-context-learning. https:\/\/github.com\/EgoAlpha\/prompt-in-context-learning."},{"key":"e_1_3_2_1_31_1","unstructured":"Kawin Ethayarajh Heidi Zhang Yizhong Wang and Dan Jurafsky. 2023. Stanford human preferences dataset."},{"key":"e_1_3_2_1_32_1","unstructured":"Deep Ganguli Liane Lovitt Jackson Kernion Amanda Askell Yuntao Bai Saurav Kadavath Ben Mann Ethan Perez Nicholas Schiefer Kamal Ndousse et al. 2022. Red teaming language models to reduce harms: Methods scaling behaviors and lessons learned. arXiv preprint arXiv:2209.07858 (2022)."},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"crossref","unstructured":"Balreet Grewal Wentao Lu Sarah Nadi and Cor-Paul Bezemer. 2024. Analyzing Developer Use of ChatGPT Generated Code in Open Source GitHub Projects. (2024).","DOI":"10.1145\/3643991.3645072"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1145\/3597503.3623306"},{"key":"e_1_3_2_1_35_1","unstructured":"Hamish Ivison Yizhong Wang Valentina Pyatkin Nathan Lambert Matthew Peters Pradeep Dasigi Joel Jang David Wadden Noah A. Smith Iz Beltagy and Hannaneh Hajishirzi. 2023. Camels in a Changing Climate: Enhancing LM Adaptation with Tulu 2. arXiv:2311.10702 [cs.CL]"},{"key":"e_1_3_2_1_36_1","volume-title":"Beavertails: Towards improved safety alignment of llm via a human-preference dataset. Advances in Neural Information Processing Systems 36","author":"Ji Jiaming","year":"2024","unstructured":"Jiaming Ji, Mickel Liu, Josef Dai, Xuehai Pan, Chi Zhang, Ce Bian, Boyuan Chen, Ruiyang Sun, Yizhou Wang, and Yaodong Yang. 2024. Beavertails: Towards improved safety alignment of llm via a human-preference dataset. Advances in Neural Information Processing Systems 36 (2024)."},{"key":"e_1_3_2_1_37_1","unstructured":"Secure Learning Lab. 2024. LLM Safety Leaderboard. https:\/\/huggingface.co\/spaces\/AI-Secure\/llm-trustworthy-leaderboard"},{"key":"e_1_3_2_1_38_1","volume-title":"International Conference on Machine Learning. PMLR","author":"Lai Yuhang","year":"2023","unstructured":"Yuhang Lai, Chengxi Li, Yiming Wang, Tianyi Zhang, Ruiqi Zhong, Luke Zettlemoyer, Wen-tau Yih, Daniel Fried, Sida Wang, and Tao Yu. 2023. DS-1000: A natural and reliable benchmark for data science code generation. In International Conference on Machine Learning. PMLR, 18319--18345."},{"key":"e_1_3_2_1_39_1","volume-title":"Yangtian Zi, Niklas Muennighoff, Denis Kocetkov, Chenghao Mou, Marc Marone, Christopher Akiki, Jia Li, Jenny Chim, et al.","author":"Li Raymond","year":"2023","unstructured":"Raymond Li, Loubna Ben Allal, Yangtian Zi, Niklas Muennighoff, Denis Kocetkov, Chenghao Mou, Marc Marone, Christopher Akiki, Jia Li, Jenny Chim, et al. 2023. Starcoder: may the source be with you! arXiv preprint arXiv:2305.06161 (2023)."},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"crossref","unstructured":"Zi Lin Zihan Wang Yongqi Tong Yangkun Wang Yuxin Guo Yujia Wang and Jingbo Shang. 2023. ToxicChat: Unveiling Hidden Challenges of Toxicity Detection in Real-World User-AI Conversation. arXiv:2310.17389 [cs.CL]","DOI":"10.18653\/v1\/2023.findings-emnlp.311"},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1145\/3324884.3416591"},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1162\/tacl_a_00638"},{"key":"e_1_3_2_1_43_1","first-page":"1","article-title":"Pre-train, prompt, and predict: A systematic survey of prompting methods in natural language processing","volume":"55","author":"Liu Pengfei","year":"2023","unstructured":"Pengfei Liu, Weizhe Yuan, Jinlan Fu, Zhengbao Jiang, Hiroaki Hayashi, and Graham Neubig. 2023. Pre-train, prompt, and predict: A systematic survey of prompting methods in natural language processing. Comput. Surveys 55, 9 (2023), 1--35.","journal-title":"Comput. Surveys"},{"key":"e_1_3_2_1_44_1","volume-title":"An Empirical Study on Low Code Programming using Traditional vs Large Language Model Support. arXiv preprint arXiv:2402.01156","author":"Liu Yongkun","year":"2024","unstructured":"Yongkun Liu, Jiachi Chen, Tingting Bi, John Grundy, Yanlin Wang, Ting Chen, Yutian Tang, and Zibin Zheng. 2024. An Empirical Study on Low Code Programming using Traditional vs Large Language Model Support. arXiv preprint arXiv:2402.01156 (2024)."},{"key":"e_1_3_2_1_45_1","volume-title":"Jailbreaking chatgpt via prompt engineering: An empirical study. arXiv preprint arXiv:2305.13860","author":"Liu Yi","year":"2023","unstructured":"Yi Liu, Gelei Deng, Zhengzi Xu, Yuekang Li, Yaowen Zheng, Ying Zhang, Lida Zhao, Tianwei Zhang, and Yang Liu. 2023. Jailbreaking chatgpt via prompt engineering: An empirical study. arXiv preprint arXiv:2305.13860 (2023)."},{"key":"e_1_3_2_1_46_1","volume-title":"Codexglue: A machine learning benchmark dataset for code understanding and generation. arXiv preprint arXiv:2102.04664","author":"Lu Shuai","year":"2021","unstructured":"Shuai Lu, Daya Guo, Shuo Ren, Junjie Huang, Alexey Svyatkovskiy, Ambrosio Blanco, Colin Clement, Dawn Drain, Daxin Jiang, Duyu Tang, et al. 2021. Codexglue: A machine learning benchmark dataset for code understanding and generation. arXiv preprint arXiv:2102.04664 (2021)."},{"key":"e_1_3_2_1_47_1","unstructured":"Mantas Mazeika Long Phan Xuwang Yin Andy Zou Zifan Wang Norman Mu Elham Sakhaee Nathaniel Li Steven Basart Bo Li David Forsyth and Dan Hendrycks. 2024. HarmBench: A Standardized Evaluation Framework for Automated Red Teaming and Robust Refusal. (2024). arXiv:2402.04249 [cs.LG]"},{"key":"e_1_3_2_1_48_1","unstructured":"Microsoft. 2024. what-is-malware. https:\/\/www.microsoft.com\/zh-cn\/security\/business\/security-101\/what-is-malware"},{"key":"e_1_3_2_1_49_1","unstructured":"Kaiwen Ning Jiachi Chen Qingyuan Zhong Tao Zhang Yanlin Wang Wei Li Yu Zhang Weizhe Zhang and Zibin Zheng. 2024. MCGMark: An Encodable and Robust Online Watermark for LLM-Generated Malicious Code. arXiv:2408.01354 [cs.CR] https:\/\/arxiv.org\/abs\/2408.01354"},{"key":"e_1_3_2_1_50_1","unstructured":"OpenAI. 2024. Openai api interface. (2024). https:\/\/platform.openai.com\/docs\/api-reference"},{"key":"e_1_3_2_1_51_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2005.1416462"},{"key":"e_1_3_2_1_52_1","unstructured":"Long Ouyang Jeffrey Wu Xu Jiang Diogo Almeida Carroll Wainwright Pamela Mishkin Chong Zhang Sandhini Agarwal Katarina Slama Alex Ray et al. 2022. Training language models to follow instructions with human feedback. Advances in neural information processing systems 35 (2022) 27730--27744."},{"key":"e_1_3_2_1_53_1","volume-title":"Phu Mon Htut, and Samuel R Bowman","author":"Parrish Alicia","year":"2021","unstructured":"Alicia Parrish, Angelica Chen, Nikita Nangia, Vishakh Padmakumar, Jason Phang, Jana Thompson, Phu Mon Htut, and Samuel R Bowman. 2021. BBQ: A hand-built bias benchmark for question answering. arXiv preprint arXiv:2110.08193 (2021)."},{"key":"e_1_3_2_1_54_1","doi-asserted-by":"publisher","DOI":"10.1109\/SP46215.2023.10179420"},{"key":"e_1_3_2_1_55_1","volume-title":"Hakan Gul, Yiming Tang, Weiyi Shang, and Zhe Yu.","author":"Reddy Puttaparthi Poorna Chander","year":"2023","unstructured":"Poorna Chander Reddy Puttaparthi, Soham Sanjay Deo, Hakan Gul, Yiming Tang, Weiyi Shang, and Zhe Yu. 2023. Comprehensive evaluation of chatgpt reliability through multilingual inquiries. arXiv preprint arXiv:2312.10524 (2023)."},{"key":"e_1_3_2_1_56_1","volume-title":"Direct preference optimization: Your language model is secretly a reward model. Advances in Neural Information Processing Systems 36","author":"Rafailov Rafael","year":"2024","unstructured":"Rafael Rafailov, Archit Sharma, Eric Mitchell, Christopher D Manning, Stefano Ermon, and Chelsea Finn. 2024. Direct preference optimization: Your language model is secretly a reward model. Advances in Neural Information Processing Systems 36 (2024)."},{"key":"e_1_3_2_1_57_1","volume-title":"SafetyPrompts: a Systematic Review of Open Datasets for Evaluating and Improving Large Language Model Safety. arXiv preprint arXiv:2404.05399","author":"R\u00f6ttger Paul","year":"2024","unstructured":"Paul R\u00f6ttger, Fabio Pernisi, Bertie Vidgen, and Dirk Hovy. 2024. SafetyPrompts: a Systematic Review of Open Datasets for Evaluating and Improving Large Language Model Safety. arXiv preprint arXiv:2404.05399 (2024)."},{"key":"e_1_3_2_1_58_1","volume-title":"Yossi Adi, Jingyu Liu, Tal Remez, J\u00e9r\u00e9my Rapin, et al.","author":"Roziere Baptiste","year":"2023","unstructured":"Baptiste Roziere, Jonas Gehring, Fabian Gloeckle, Sten Sootla, Itai Gat, Xiaoqing Ellen Tan, Yossi Adi, Jingyu Liu, Tal Remez, J\u00e9r\u00e9my Rapin, et al. 2023. Code llama: Open foundation models for code. arXiv preprint arXiv:2308.12950 (2023)."},{"key":"e_1_3_2_1_59_1","unstructured":"Nino Scherrer Claudia Shi Amir Feder and David Blei. 2023. Evaluating the Moral Beliefs Encoded in LLMs."},{"key":"e_1_3_2_1_60_1","volume-title":"Large language model alignment: A survey. arXiv preprint arXiv:2309.15025","author":"Shen Tianhao","year":"2023","unstructured":"Tianhao Shen, Renren Jin, Yufei Huang, Chuang Liu, Weilong Dong, Zishan Guo, Xinwei Wu, Yan Liu, and Deyi Xiong. 2023. Large language model alignment: A survey. arXiv preprint arXiv:2309.15025 (2023)."},{"key":"e_1_3_2_1_61_1","unstructured":"MosaicML NLP Team. 2023. Introducing MPT-7B: A New Standard for Open-Source Commercially Usable LLMs. www.mosaicml.com\/blog\/mpt-7b Accessed: 2023-03-28."},{"key":"e_1_3_2_1_62_1","unstructured":"Hugo Touvron Louis Martin Kevin Stone Peter Albert Amjad Almahairi Yasmine Babaei Nikolay Bashlykov Soumya Batra Prajjwal Bhargava Shruti Bhosale et al. 2023. Llama 2: Open foundation and fine-tuned chat models. arXiv preprint arXiv:2307.09288 (2023)."},{"key":"e_1_3_2_1_63_1","volume-title":"Zephyr: Direct Distillation of LM Alignment. arXiv:2310.16944 [cs.LG]","author":"Tunstall Lewis","year":"2023","unstructured":"Lewis Tunstall, Edward Beeching, Nathan Lambert, Nazneen Rajani, Kashif Rasul, Younes Belkada, Shengyi Huang, Leandro von Werra, Cl\u00e9mentine Fourrier, Nathan Habib, Nathan Sarrazin, Omar Sanseviero, Alexander M. Rush, and Thomas Wolf. 2023. Zephyr: Direct Distillation of LM Alignment. arXiv:2310.16944 [cs.LG]"},{"key":"e_1_3_2_1_64_1","volume-title":"Attention is all you need. Advances in neural information processing systems 30","author":"Vaswani Ashish","year":"2017","unstructured":"Ashish Vaswani, Noam Shazeer, Niki Parmar, Jakob Uszkoreit, Llion Jones, Aidan N Gomez, \u0141ukasz Kaiser, and Illia Polosukhin. 2017. Attention is all you need. Advances in neural information processing systems 30 (2017)."},{"key":"e_1_3_2_1_65_1","volume-title":"Decodingtrust: A comprehensive assessment of trustworthiness in gpt models. arXiv preprint arXiv:2306.11698","author":"Wang Boxin","year":"2023","unstructured":"Boxin Wang, Weixin Chen, Hengzhi Pei, Chulin Xie, Mintong Kang, Chenhui Zhang, Chejian Xu, Zidi Xiong, Ritik Dutta, Rylan Schaeffer, et al. 2023. Decodingtrust: A comprehensive assessment of trustworthiness in gpt models. arXiv preprint arXiv:2306.11698 (2023)."},{"key":"e_1_3_2_1_66_1","unstructured":"Yanlin Wang Yanli Wang Daya Guo Jiachi Chen Ruikai Zhang Yuchi Ma and Zibin Zheng. 2024. RLCoder: Reinforcement Learning for Repository-Level Code Completion. arXiv:2407.19487 [cs.SE] https:\/\/arxiv.org\/abs\/2407.19487"},{"key":"e_1_3_2_1_67_1","volume-title":"Jailbroken: How does llm safety training fail? Advances in Neural Information Processing Systems 36","author":"Wei Alexander","year":"2024","unstructured":"Alexander Wei, Nika Haghtalab, and Jacob Steinhardt. 2024. Jailbroken: How does llm safety training fail? Advances in Neural Information Processing Systems 36 (2024)."},{"key":"e_1_3_2_1_68_1","volume-title":"Denny Zhou, et al.","author":"Wei Jason","year":"2022","unstructured":"Jason Wei, Xuezhi Wang, Dale Schuurmans, Maarten Bosma, Fei Xia, Ed Chi, Quoc V Le, Denny Zhou, et al. 2022. Chain-of-thought prompting elicits reasoning in large language models. Advances in neural information processing systems 35 (2022), 24824--24837."},{"key":"e_1_3_2_1_69_1","volume-title":"Hyperion: Unveiling DApp Inconsistencies using LLM and Dataflow-Guided Symbolic Execution. arXiv:2408.06037 [cs.SE] https:\/\/arxiv.org\/abs\/2408.06037","author":"Yang Shuo","year":"2024","unstructured":"Shuo Yang, Xingwei Lin, Jiachi Chen, Qingyuan Zhong, Lei Xiao, Renke Huang, Yanlin Wang, and Zibin Zheng. 2024. Hyperion: Unveiling DApp Inconsistencies using LLM and Dataflow-Guided Symbolic Execution. arXiv:2408.06037 [cs.SE] https:\/\/arxiv.org\/abs\/2408.06037"},{"key":"e_1_3_2_1_70_1","doi-asserted-by":"publisher","DOI":"10.1145\/3597503.3623316"},{"key":"e_1_3_2_1_71_1","volume-title":"Gptfuzzer: Red teaming large language models with auto-generated jailbreak prompts. arXiv preprint arXiv:2309.10253","author":"Yu Jiahao","year":"2023","unstructured":"Jiahao Yu, Xingwei Lin, and Xinyu Xing. 2023. Gptfuzzer: Red teaming large language models with auto-generated jailbreak prompts. arXiv preprint arXiv:2309.10253 (2023)."},{"key":"e_1_3_2_1_72_1","volume-title":"Repocoder: Repository-level code completion through iterative retrieval and generation. arXiv preprint arXiv:2303.12570","author":"Zhang Fengji","year":"2023","unstructured":"Fengji Zhang, Bei Chen, Yue Zhang, Jacky Keung, Jin Liu, Daoguang Zan, Yi Mao, Jian-Guang Lou, and Weizhu Chen. 2023. Repocoder: Repository-level code completion through iterative retrieval and generation. arXiv preprint arXiv:2303.12570 (2023)."},{"key":"e_1_3_2_1_73_1","volume-title":"Safetybench: Evaluating the safety of large language models with multiple choice questions. arXiv preprint arXiv:2309.07045","author":"Zhang Zhexin","year":"2023","unstructured":"Zhexin Zhang, Leqi Lei, Lindong Wu, Rui Sun, Yongkang Huang, Chong Long, Xiao Liu, Xuanyu Lei, Jie Tang, and Minlie Huang. 2023. Safetybench: Evaluating the safety of large language models with multiple choice questions. arXiv preprint arXiv:2309.07045 (2023)."},{"key":"e_1_3_2_1_74_1","volume-title":"A Survey of Large Language Models. arXiv preprint arXiv:2303.18223","author":"Zhao Wayne Xin","year":"2023","unstructured":"Wayne Xin Zhao, Kun Zhou, Junyi Li, Tianyi Tang, Xiaolei Wang, Yupeng Hou, Yingqian Min, Beichen Zhang, Junjie Zhang, Zican Dong, Yifan Du, Chen Yang, Yushuo Chen, Zhipeng Chen, Jinhao Jiang, Ruiyang Ren, Yifan Li, Xinyu Tang, Zikang Liu, Peiyu Liu, Jian-Yun Nie, and Ji-Rong Wen. 2023. A Survey of Large Language Models. arXiv preprint arXiv:2303.18223 (2023). http:\/\/arxiv.org\/abs\/2303.18223"},{"key":"e_1_3_2_1_75_1","unstructured":"Lianmin Zheng Wei-Lin Chiang Ying Sheng Siyuan Zhuang Zhanghao Wu Yonghao Zhuang Zi Lin Zhuohan Li Dacheng Li Eric Xing et al. 2024. Judging llm-as-a-judge with mt-bench and chatbot arena. Advances in Neural Information Processing Systems 36 (2024)."},{"key":"e_1_3_2_1_76_1","volume-title":"Towards an understanding of large language models in software engineering tasks. arXiv preprint arXiv:2308.11396","author":"Zheng Zibin","year":"2023","unstructured":"Zibin Zheng, Kaiwen Ning, Jiachi Chen, Yanlin Wang, Wenqing Chen, Lianghong Guo, and Weicheng Wang. 2023. Towards an understanding of large language models in software engineering tasks. arXiv preprint arXiv:2308.11396 (2023)."},{"key":"e_1_3_2_1_77_1","volume-title":"A survey of large language models for code: Evolution, benchmarking, and future trends. arXiv preprint arXiv:2311.10372","author":"Zheng Zibin","year":"2023","unstructured":"Zibin Zheng, Kaiwen Ning, Yanlin Wang, Jingwen Zhang, Dewu Zheng, Mingxi Ye, and Jiachi Chen. 2023. A survey of large language models for code: Evolution, benchmarking, and future trends. arXiv preprint arXiv:2311.10372 (2023)."},{"key":"e_1_3_2_1_78_1","first-page":"2232","article-title":"ICE-Score: Instructing Large Language Models to Evaluate Code","volume":"2024","author":"Zhuo Terry Yue","year":"2024","unstructured":"Terry Yue Zhuo. 2024. ICE-Score: Instructing Large Language Models to Evaluate Code. In Findings of the Association for Computational Linguistics: EACL 2024. 2232--2242.","journal-title":"Findings of the Association for Computational Linguistics: EACL"}],"event":{"name":"ASE '24: 39th IEEE\/ACM International Conference on Automated Software Engineering","location":"Sacramento CA USA","acronym":"ASE '24","sponsor":["SIGAI ACM Special Interest Group on Artificial Intelligence","SIGSOFT ACM Special Interest Group on Software Engineering","IEEE CS"]},"container-title":["Proceedings of the 39th IEEE\/ACM International Conference on Automated Software Engineering"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3691620.3695480","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3691620.3695480","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T00:06:19Z","timestamp":1750291579000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3691620.3695480"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,27]]},"references-count":78,"alternative-id":["10.1145\/3691620.3695480","10.1145\/3691620"],"URL":"https:\/\/doi.org\/10.1145\/3691620.3695480","relation":{},"subject":[],"published":{"date-parts":[[2024,10,27]]},"assertion":[{"value":"2024-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}