{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,6]],"date-time":"2026-05-06T23:16:20Z","timestamp":1778109380152,"version":"3.51.4"},"publisher-location":"Cham","reference-count":38,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783032160911","type":"print"},{"value":"9783032160928","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026]]},"DOI":"10.1007\/978-3-032-16092-8_19","type":"book-chapter","created":{"date-parts":[[2026,5,6]],"date-time":"2026-05-06T23:02:11Z","timestamp":1778108531000},"page":"346-365","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Analysing Safety Risks in\u00a0LLMs Fine-Tuned with\u00a0Pseudo-malicious Cyber Security Data"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-5402-7837","authenticated-orcid":false,"given":"Adel","family":"ElZemity","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1830-1587","authenticated-orcid":false,"given":"Budi","family":"Arief","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5628-7328","authenticated-orcid":false,"given":"Shujun","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2026,5,1]]},"reference":[{"key":"19_CR1","doi-asserted-by":"crossref","unstructured":"Alotaibi, L., Seher, S., Mohammad, N.: Cyberattacks using ChatGPT: Exploring malicious content generation through prompt engineering. Proceedings of the 2024 ASU International Conference in Emerging Technologies for Sustainability and Intelligent Systems pp. 1304\u20131311 (2024)","DOI":"10.1109\/ICETSIS61505.2024.10459698"},{"key":"19_CR2","unstructured":"Arrieta, A., Ugarte, M., Valle, P., Parejo, J.A., Segura, S.: o3-mini vs deepseek-r1: Which one is safer? (2025), https:\/\/arxiv.org\/abs\/2501.18438"},{"key":"19_CR3","unstructured":"Bianchi, F., Suzgun, M., Attanasio, G., Rottger, P., Jurafsky, D., Hashimoto, T., Zou, J.: Safety-tuned LLaMAs: Lessons from improving the safety of large language models that follow instructions. In: The Twelfth International Conference on Learning Representations (2024)"},{"issue":"9","key":"19_CR4","doi-asserted-by":"publisher","first-page":"1163","DOI":"10.3897\/jucs.134739","volume":"30","author":"O \u00c7etin","year":"2024","unstructured":"\u00c7etin, O., Ekmekcioglu, E., Arief, B., Hernandez-Castro, J.: An Empirical Evaluation of Large Language Models in Static Code Analysis for PHP Vulnerability Detection. J. of Universal Comp. Sci. 30(9), 1163\u20131183 (2024)","journal-title":"J. of Universal Comp. Sci."},{"key":"19_CR5","doi-asserted-by":"publisher","unstructured":"Charan, P.V.S., Chunduri, H., Anand, P.M., Shukla, S.K.: From text to MITRE techniques: Exploring the malicious use of large language models for generating cyber attack payloads (2023). https:\/\/doi.org\/10.48550\/arXiv.2305.15336","DOI":"10.48550\/arXiv.2305.15336"},{"key":"19_CR6","unstructured":"Chen, K., Wang, C., Yang, K., Han, J., Hong, L., Mi, F., Xu, H., Liu, Z., Huang, W., Li, Z., Yeung, D.Y., Shang, L.: Gaining wisdom from setbacks: Aligning large language models via mistake analysis. In: The Twelfth International Conference on Learning Representations (2024)"},{"key":"19_CR7","doi-asserted-by":"crossref","unstructured":"Chen, S., Zharmagambetov, A., Mahloujifar, S., Chaudhuri, K., Wagner, D., Guo, C.: SecAlign: Defending Against Prompt Injection with Preference Optimization (2025), https:\/\/arxiv.org\/abs\/2410.05451","DOI":"10.1145\/3719027.3744836"},{"key":"19_CR8","unstructured":"Choi, H.K., Du, X., Li, Y.: Safety-Aware Fine-Tuning of Large Language Models. In: Neurips Safe Generative AI Workshop 2024 (2024)"},{"key":"19_CR9","unstructured":"Confident AI: DeepEval: The LLM Evaluation Framework (2024), https:\/\/github.com\/confident-ai\/deepeval, accessed: 2024-12-04"},{"key":"19_CR10","unstructured":"Confident AI: DeepEval: The Open-Source LLM Evaluation Framework (2024), https:\/\/docs.confident-ai.com\/docs\/red-teaming-introduction"},{"key":"19_CR11","unstructured":"DeepSeek-AI, Guo, D., Yang, D., et\u00a0al.: Deepseek-r1: Incentivizing reasoning capability in llms via reinforcement learning (2025), https:\/\/arxiv.org\/abs\/2501.12948"},{"key":"19_CR12","unstructured":"Derczynski, L., Galinkin, E., Martin, J., Majumdar, S., Inie, N.: garak: A Framework for Security Probing Large Language Models. https:\/\/garak.ai (2024)"},{"key":"19_CR13","doi-asserted-by":"publisher","first-page":"126176","DOI":"10.1109\/ACCESS.2024.3450388","volume":"12","author":"E Derner","year":"2023","unstructured":"Derner, E., Batistic, K., Zah\u00e1lka, J., Babu\u0161ka, R.: A security risk taxonomy for prompt-based interaction with large language models. IEEE Access 12, 126176\u2013126187 (2023)","journal-title":"IEEE Access"},{"key":"19_CR14","unstructured":"Eiras, F., Petrov, A., Torr, P., Kumar, M.P., Bibi, A.: Mimicking user data: On mitigating fine-tuning risks in closed large language models. In: ICML 2024 Next Generation of AI Safety Workshop (2024)"},{"key":"19_CR15","doi-asserted-by":"crossref","unstructured":"ElZemity, A., Arief, B., Li, S.: CyberLLMInstruct: A pseudo-malicious dataset revealing safety-performance trade-offs in cyber security LLM fine-tuning. In: Proceedings of the 2025 Workshop on Artificial Intelligence and Security. ACM, New York, NY, USA (2025), preprint available at https:\/\/doi.org\/10.48550\/arXiv.2503.09334","DOI":"10.1145\/3733799.3762968"},{"key":"19_CR16","doi-asserted-by":"crossref","unstructured":"Falade, P.V.: Decoding the Threat Landscape: ChatGPT, FraudGPT, and WormGPT in Social Engineering Attacks (2023), https:\/\/arxiv.org\/abs\/2310.05595","DOI":"10.32628\/CSEIT2390533"},{"key":"19_CR17","unstructured":"Firdhous, M.F.M., Elbreiki, W., Abdullahi, I., Sudantha, B.H., Budiarto, R.: WormGPT: A large language model chatbot for criminals. In: Proceedings of the 2023 24th International Arab Conference on Information Technology (2023)"},{"key":"19_CR18","unstructured":"Google AI: Gemma 2 9B Model (2024), https:\/\/huggingface.co\/google\/gemma-2-9b, accessed: 2024-10-27"},{"key":"19_CR19","unstructured":"Hendrycks, D., Burns, C., Basart, S., Zou, A., Mazeika, M., Song, D., Steinhardt, J.: Measuring massive multitask language understanding (2021), https:\/\/arxiv.org\/abs\/2009.03300"},{"key":"19_CR20","unstructured":"Hsu, C.Y., Tsai, Y.L., Lin, C.H., Chen, P.Y., Yu, C.M., Huang, C.Y.: Safe loRA: The silver lining of reducing safety risks when finetuning large language models. In: The Thirty-eighth Annual Conference on Neural Information Processing Systems (2024)"},{"key":"19_CR21","unstructured":"Jain, S., Lubana, E.S., Oksuz, K., Joy, T., Torr, P., Sanyal, A., Dokania, P.K.: What Makes and Breaks Safety Fine-tuning? A Mechanistic Study. In: The Thirty-eighth Annual Conference on Neural Information Processing Systems (2024)"},{"key":"19_CR22","unstructured":"Longpre, S., Hou, L., Vu, T., Webson, A., Chung, H.W., Tay, Y., Zhou, D., Le, Q.V., Zoph, B., Wei, J., Roberts, A.: The flan collection: Designing data and methods for effective instruction tuning. In: Krause, A., Brunskill, E., Cho, K., Engelhardt, B., Sabato, S., Scarlett, J. (eds.) Proceedings of the 40th International Conference on Machine Learning. vol.\u00a0202, pp. 22631\u201322648. PMLR (23\u201329 Jul 2023)"},{"key":"19_CR23","unstructured":"Meta AI: Llama 3 8B Model (2024), https:\/\/huggingface.co\/meta-llama\/Meta-Llama-3-8B, accessed: 2024-10-27"},{"key":"19_CR24","unstructured":"Mistral AI: Mistral 7B Model (2024), https:\/\/huggingface.co\/mistralai\/Mistral-7B-v0.3, accessed: 2024-10-27"},{"key":"19_CR25","unstructured":"OWASP Foundation: OWASP top 10 for large language model applications (2025), https:\/\/owasp.org\/www-project-top-10-for-large-language-model-applications\/, accessed: 2024-12-16"},{"key":"19_CR26","doi-asserted-by":"crossref","unstructured":"Ozturk, O.S., Ekmekcioglu, E., Cetin, O., Arief, B., Hernandez-Castro, J.: New tricks to old codes: can ai chatbots replace static code analysis tools? In: Proceedings of the 2023 European Interdisciplinary Cybersecurity Conference. pp. 13\u201318 (2023)","DOI":"10.1145\/3590777.3590780"},{"key":"19_CR27","unstructured":"Peng, S., Chen, P.Y., Hull, M.D., Chau, D.H.: Navigating the Safety Landscape: Measuring Risks in Finetuning Large Language Models. In: The Thirty-eighth Annual Conference on Neural Information Processing Systems (2024)"},{"key":"19_CR28","unstructured":"Qi, X., Zeng, Y., Xie, T., Chen, P.Y., Jia, R., Mittal, P., Henderson, P.: Fine-tuning Aligned Language Models Compromises Safety, Even When Users Do Not Intend To! In: The Twelfth International Conference on Learning Representations (2024)"},{"key":"19_CR29","doi-asserted-by":"publisher","first-page":"26839","DOI":"10.1109\/ACCESS.2024.3365742","volume":"12","author":"MAK Raiaan","year":"2024","unstructured":"Raiaan, M.A.K., Mukta, M.S.H., Fatema, K., et al.: A review on large language models: Architectures, applications, taxonomies, open issues and challenges. IEEE Access 12, 26839\u201326874 (2024)","journal-title":"IEEE Access"},{"key":"19_CR30","unstructured":"robomotic: LLM Guardrails Frameworks (2025), https:\/\/github.com\/robomotic\/awesome-guide-ai-safety\/blob\/master\/TOOLS.md, accessed: 2025-04-30"},{"key":"19_CR31","doi-asserted-by":"crossref","unstructured":"Roy, S.S., Thota, P., Naragam, K.V., Nilizadeh, S.: From chatbots to phishbots?: Phishing scam generation in commercial large language models. In: Proceedings of the 2024 IEEE Symposium on Security and Privacy. pp. 36\u201354 (2024)","DOI":"10.1109\/SP54263.2024.00182"},{"key":"19_CR32","doi-asserted-by":"publisher","first-page":"72303","DOI":"10.1109\/ACCESS.2024.3403858","volume":"12","author":"Z S\u00e1godi","year":"2024","unstructured":"S\u00e1godi, Z., Siket, I., Ferenc, R.: Methodology for code synthesis evaluation of LLMs presented by a case study of ChatGPT and Copilot. IEEE Access 12, 72303\u201372316 (2024)","journal-title":"IEEE Access"},{"key":"19_CR33","unstructured":"Sun, J., Shaib, C., Wallace, B.C.: Evaluating the zero-shot robustness of instruction-tuned language models. In: The Twelfth International Conference on Learning Representations (2024)"},{"issue":"6","key":"19_CR34","first-page":"7","volume":"3","author":"R Taori","year":"2023","unstructured":"Taori, R., Gulrajani, I., Zhang, T., Dubois, Y., Li, X., Guestrin, C., Liang, P., Hashimoto, T.B.: Alpaca: A strong, replicable instruction-following model. Stanford Center for Research on Foundation Models 3(6), 7 (2023)","journal-title":"Stanford Center for Research on Foundation Models"},{"key":"19_CR35","unstructured":"von Werra, L., Belkada, Y., Tunstall, L., Beeching, E., Thrush, T., Lambert, N., Huang, S., Rasul, K., Gallou\u00e9dec, Q.: Trl: Transformer reinforcement learning. https:\/\/github.com\/huggingface\/trl (2020)"},{"key":"19_CR36","doi-asserted-by":"crossref","unstructured":"Wolf, T., Debut, L., Sanh, V., et\u00a0al.: Transformers: State-of-the-art natural language processing. In: Proceedings of the 2020 Conference on Empirical Methods in Natural Language Processing: System Demonstrations. pp. 38\u201345. ACL (2020)","DOI":"10.18653\/v1\/2020.emnlp-demos.6"},{"key":"19_CR37","doi-asserted-by":"crossref","unstructured":"Yao, Y., Duan, J., Xu, K., Cai, Y., Sun, Z., Zhang, Y.: A survey on large language model (LLM) security and privacy: The good, the bad, and the ugly. High-Confidence Computing 4(2), 100211:1\u2013100211:21 (2024)","DOI":"10.1016\/j.hcc.2024.100211"},{"key":"19_CR38","unstructured":"Zhu, M., Yang, L., Wei, Y., Zhang, N., Zhang, Y.: Locking down the finetuned llms safety (2024), https:\/\/arxiv.org\/abs\/2410.10343"}],"container-title":["Lecture Notes in Computer Science","Computer Security. ESORICS 2025 International Workshops"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-032-16092-8_19","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,5,6]],"date-time":"2026-05-06T23:02:19Z","timestamp":1778108539000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-032-16092-8_19"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026]]},"ISBN":["9783032160911","9783032160928"],"references-count":38,"URL":"https:\/\/doi.org\/10.1007\/978-3-032-16092-8_19","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026]]},"assertion":[{"value":"1 May 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"The authors have no competing interests to declare that are relevant to the content of this article.","order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Disclosure of Interests"}},{"value":"ESORICS","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Symposium on Research in Computer Security","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Toulouse","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"France","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2025","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"22 September 2025","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"26 September 2025","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"30","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"esorics2025","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/www.esorics2025.org\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}