{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,11]],"date-time":"2026-07-11T08:11:55Z","timestamp":1783757515019,"version":"3.55.0"},"reference-count":195,"publisher":"Springer Science and Business Media LLC","issue":"6","license":[{"start":{"date-parts":[[2026,5,23]],"date-time":"2026-05-23T00:00:00Z","timestamp":1779494400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,5,23]],"date-time":"2026-05-23T00:00:00Z","timestamp":1779494400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62276128"],"award-info":[{"award-number":["62276128"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62192783"],"award-info":[{"award-number":["62192783"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100004608","name":"Natural Science Foundation of Jiangsu Province","doi-asserted-by":"publisher","award":["BK20243051"],"award-info":[{"award-number":["BK20243051"]}],"id":[{"id":"10.13039\/501100004608","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Jiangsu Science and Technology Major Project","award":["BG2024031"],"award-info":[{"award-number":["BG2024031"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Mach Learn"],"published-print":{"date-parts":[[2026,6]]},"DOI":"10.1007\/s10994-025-06930-x","type":"journal-article","created":{"date-parts":[[2026,5,23]],"date-time":"2026-05-23T08:38:08Z","timestamp":1779525488000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Mitigating Security Risks in Large Language Models: A Full Lifecycle Perspective"],"prefix":"10.1007","volume":"115","author":[{"given":"Yanming","family":"Wang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zhixin","family":"Bai","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jing","family":"Huo","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Boyan","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jiaheng","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Hongye","family":"Cao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Fanyu","family":"Meng","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xi","family":"Yang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yang","family":"Gao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,5,23]]},"reference":[{"key":"6930_CR1","doi-asserted-by":"crossref","unstructured":"Abadi, M., Chu, A., Goodfellow, I., McMahan, H.B., Mironov, I., Talwar, K., & Zhang, L. (2016). Deep learning with differential privacy. In Proceedings of the 2016 ACM SIGSAC conference on computer and communications security (pp. 308\u2013318).","DOI":"10.1145\/2976749.2978318"},{"key":"6930_CR2","doi-asserted-by":"crossref","unstructured":"Abid, A., Farooqi, M., & Zou, J. (2021). Persistent anti-muslim bias in large language models. In Proceedings of the 2021 AAAI\/ACM conference on AI, ethics, and society (pp. 298\u2013306).","DOI":"10.1145\/3461702.3462624"},{"issue":"4","key":"6930_CR3","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3214303","volume":"51","author":"A Acar","year":"2018","unstructured":"Acar, A., Aksu, H., Uluagac, A. S., & Conti, M. (2018). A survey on homomorphic encryption schemes: Theory and implementation. ACM Computing Surveys (CSUR), 51(4), 1\u201335.","journal-title":"ACM Computing Surveys (CSUR)"},{"issue":"1","key":"6930_CR4","first-page":"1","volume":"10","author":"SF Ahmad","year":"2023","unstructured":"Ahmad, S. F., Han, H., Alam, M. M., Rehmat, M., Irshad, M., Arra\u00f1o-Mu\u00f1oz, M., Ariza-Montes, A., et al. (2023). Impact of artificial intelligence on human loss in decision making, laziness and safety in education. Humanities and Social Sciences Communications, 10(1), 1\u201314.","journal-title":"Humanities and Social Sciences Communications"},{"key":"6930_CR5","unstructured":"Alon, G., & Kamfonas, M. (2023). Detecting language model attacks with perplexity. arXiv preprint arXiv:2308.14132"},{"key":"6930_CR6","unstructured":"Azar, M.G., Guo, Z.D., Piot, B., Munos, R., Rowland, M., Valko, M., & Calandriello, D. (2024). A general theoretical paradigm to understand learning from human preferences. In: International Conference on Artificial Intelligence and Statistics, 4447\u20134455. PMLR"},{"key":"6930_CR7","unstructured":"Bai, Y., Jones, A., Ndousse, K., Askell, A., Chen, A., DasSarma, N., Drain, D., Fort, S., Ganguli, D., Henighan, T., et al. (2022). Training a helpful and harmless assistant with reinforcement learning from human feedback. arXiv preprint arXiv:2204.05862"},{"key":"6930_CR8","unstructured":"Bai, Y., Kadavath, S., Kundu, S., Askell, A., Kernion, J., Jones, A., Chen, A., Goldie, A., Mirhoseini, A., McKinnon, C., et al. (2022). Constitutional ai: Harmlessness from ai feedback. arXiv preprint arXiv:2212.08073"},{"key":"6930_CR9","doi-asserted-by":"crossref","unstructured":"Behnia, R., Ebrahimi, M.R., Pacheco, J., & Padmanabhan, B. (2022). Ew-tune: A framework for privately fine-tuning large language models with differential privacy. In: 2022 IEEE International Conference on Data Mining Workshops (ICDMW), 560\u2013566. IEEE","DOI":"10.1109\/ICDMW58026.2022.00078"},{"key":"6930_CR10","unstructured":"Bethany, M., Galiopoulos, A., Bethany, E., Karkevandi, M.B., Vishwamitra, N., & Najafirad, P. (2024). Large language model lateral spear phishing: A comparative study in large-scale organizational settings. arXiv preprint arXiv:2401.09727"},{"key":"6930_CR11","first-page":"1877","volume":"33","author":"T Brown","year":"2020","unstructured":"Brown, T., Mann, B., Ryder, N., Subbiah, M., Kaplan, J. D., Dhariwal, P., Neelakantan, A., Shyam, P., Sastry, G., Askell, A., et al. (2020). Language models are few-shot learners. Advances in Neural Information Processing Systems, 33, 1877\u20131901.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"6930_CR12","doi-asserted-by":"crossref","unstructured":"Capraro, V., Lentsch, A., Acemoglu, D., Akgun, S., Akhmedova, A., Bilancini, E., Bonnefon, J.-F., Bra\u00f1as-Garza, P., Butera, L., Douglas, K.M., et al. (2024). The impact of generative artificial intelligence on socioeconomic inequalities and policy making. PNAS nexus 3(6)","DOI":"10.1093\/pnasnexus\/pgae191"},{"key":"6930_CR13","doi-asserted-by":"publisher","unstructured":"Carlini, N., Jagielski, M., Choquette-Choo, C.A., Paleka, D., Pearce, W., Anderson, H., Terzis, A., Thomas, K., & Tramer, F. (2024). Poisoning web-scale training datasets is practical. In: 2024 IEEE Symposium on Security and Privacy (SP), pp. 407\u2013425. IEEE Computer Society, Los Alamitos, CA, USA. https:\/\/doi.org\/10.1109\/SP54263.2024.00179","DOI":"10.1109\/SP54263.2024.00179"},{"key":"6930_CR14","unstructured":"Carlini, N., Paleka, D., Dvijotham, K.D., Steinke, T., Hayase, J., Cooper, A.F., Lee, K., Jagielski, M., Nasr, M., Conmy, A., et al. (2024). Stealing part of a production language model. arXiv preprint arXiv:2403.06634"},{"key":"6930_CR15","doi-asserted-by":"crossref","unstructured":"Chang, K., Cramer, M., Soni, S., & Bamman, D. (2023). Speak, memory: An archaeology of books known to chatgpt\/gpt-4. Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing, 7312\u20137327","DOI":"10.18653\/v1\/2023.emnlp-main.453"},{"key":"6930_CR16","doi-asserted-by":"crossref","unstructured":"Chatterjee, S., Gupta, A.K., Mahor, V.K., & Sarmah, T. (2014). An efficient fine grained access control scheme based on attributes for enterprise class applications. 2014 International Conference on Signal Propagation and Computer Technology (ICSPCT 2014), 273\u2013278. IEEE","DOI":"10.1109\/ICSPCT.2014.6884907"},{"key":"6930_CR17","doi-asserted-by":"publisher","unstructured":"Chen, J., & Yang, D. (2023). Unlearn what you want to forget: Efficient unlearning for LLMs. In: Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing, pp. 12041\u201312052. Association for Computational Linguistics, Singapore. https:\/\/doi.org\/10.18653\/v1\/2023.emnlp-main.738","DOI":"10.18653\/v1\/2023.emnlp-main.738"},{"key":"6930_CR18","unstructured":"Chen, Z., Deng, Y., Yuan, H., Ji, K., & Gu, Q. (2024). Self-play fine-tuning converts weak language models to strong language models. In: Proceedings of the 41st International Conference on Machine Learning. Proceedings of Machine Learning Research, vol. 235, pp. 6621\u20136642. PMLR."},{"key":"6930_CR19","doi-asserted-by":"crossref","unstructured":"Chen, C., Feng, X., Li, Y., Lyu, L., Zhou, J., Zheng, X., & Yin, J. (2024). Integration of large language models and federated learning. Patterns 5(12)","DOI":"10.1016\/j.patter.2024.101098"},{"key":"6930_CR20","unstructured":"Chen, K., Tang, H., Liu, Q., & Xu, Y. (2025). Improved algorithms for differentially private language model alignment. arXiv preprint arXiv:2505.08849"},{"issue":"2","key":"6930_CR21","doi-asserted-by":"publisher","first-page":"48","DOI":"10.1109\/MC.2022.3218005","volume":"56","author":"P-Y Chen","year":"2023","unstructured":"Chen, P.-Y., & Das, P. (2023). Ai maintenance: a robustness perspective. Computer, 56(2), 48\u201356.","journal-title":"Computer"},{"key":"6930_CR22","unstructured":"Chowdhury, A.G., Islam, M.M., Kumar, V., Shezan, F.H., Jain, V., & Chadha, A. (2024). Breaking down the defenses: A comparative survey of attacks on large language models. arXiv preprint arXiv:2403.04786"},{"key":"6930_CR23","unstructured":"Cottier, B., Rahman, R., Fattorini, L., Maslej, N., & Owen, D. (2024). The rising costs of training frontier ai models. arXiv preprint arXiv:2405.21015"},{"key":"6930_CR24","unstructured":"Dai, J., Pan, X., Sun, R., Ji, J., Xu, X., Liu, M., Wang, Y., & Yang, Y. (2024). Safe rlhf: Safe reinforcement learning from human feedback. In: The Twelfth International Conference on Learning Representations"},{"key":"6930_CR25","doi-asserted-by":"publisher","first-page":"138872","DOI":"10.1109\/ACCESS.2019.2941376","volume":"7","author":"J Dai","year":"2019","unstructured":"Dai, J., Chen, C., & Li, Y. (2019). A backdoor attack against lstm-based text classification systems. IEEE Access, 7, 138872\u2013138878.","journal-title":"IEEE Access"},{"key":"6930_CR26","unstructured":"Das, S., Kolling, C., Khan, M.A., Amani, M., Ghosh, B., Wu, Q., Speicher, T., & Gummadi, K.P. (2025). Revisiting privacy, utility, and efficiency trade-offs when fine-tuning large language models. arXiv preprint arXiv:2502.13313"},{"key":"6930_CR27","doi-asserted-by":"publisher","unstructured":"Deng, B., Wang, W., Feng, F., Deng, Y., Wang, Q., & He, X. (2023). Attack prompt generation for red teaming and defending large language models. In: Findings of the Association for Computational Linguistics: EMNLP 2023, pp. 2176\u20132189. Association for Computational Linguistics, Singapore. https:\/\/doi.org\/10.18653\/v1\/2023.findings-emnlp.143","DOI":"10.18653\/v1\/2023.findings-emnlp.143"},{"key":"6930_CR28","unstructured":"Devlin, J. (2018). Bert: Pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805"},{"key":"6930_CR29","doi-asserted-by":"publisher","unstructured":"Ding, N., Chen, Y., Xu, B., Qin, Y., Hu, S., Liu, Z., Sun, M., & Zhou, B. (2023). Enhancing chat language models by scaling high-quality instructional conversations. In: Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing, pp. 3029\u20133051. Association for Computational Linguistics, Singapore. https:\/\/doi.org\/10.18653\/v1\/2023.emnlp-main.183","DOI":"10.18653\/v1\/2023.emnlp-main.183"},{"issue":"3","key":"6930_CR30","doi-asserted-by":"publisher","first-page":"220","DOI":"10.1038\/s42256-023-00626-4","volume":"5","author":"N Ding","year":"2023","unstructured":"Ding, N., Qin, Y., Yang, G., Wei, F., Yang, Z., Su, Y., Hu, S., Chen, Y., Chan, C.-M., Chen, W., et al. (2023). Parameter-efficient fine-tuning of large-scale pre-trained language models. Nature Machine Intelligence, 5(3), 220\u2013235.","journal-title":"Nature Machine Intelligence"},{"key":"6930_CR31","unstructured":"Dong, H., Xiong, W., Goyal, D., Zhang, Y., Chow, W., Pan, R., Diao, S., Zhang, J., SHUM, K., & Zhang, T. (2023). RAFT: Reward ranked finetuning for generative foundation model alignment. Transactions on Machine Learning Research"},{"key":"6930_CR32","doi-asserted-by":"publisher","unstructured":"Dong, Z., Zhou, Z., Yang, C., Shao, J., & Qiao, Y. (2024). Attacks, defenses and evaluations for LLM conversation safety: A survey. In: Proceedings of the 2024 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies (Volume 1: Long Papers), pp. 6734\u20136747. Association for Computational Linguistics, Mexico City, Mexico. https:\/\/doi.org\/10.18653\/v1\/2024.naacl-long.375","DOI":"10.18653\/v1\/2024.naacl-long.375"},{"key":"6930_CR33","unstructured":"Du, J., & Mi, H. (2021). Dp-fp: Differentially private forward propagation for large models. arXiv preprint arXiv:2112.14430"},{"key":"6930_CR34","unstructured":"Duan, M., Suri, A., Mireshghallah, N., Min, S., Shi, W., Zettlemoyer, L., Tsvetkov, Y., Choi, Y., Evans, D., & Hajishirzi, H. (2024). Do membership inference attacks work on large language models? In: First Conference on Language Modeling"},{"key":"6930_CR35","doi-asserted-by":"crossref","unstructured":"Dwork, C., McSherry, F., Nissim, K., & Smith, A. (2006) Calibrating noise to sensitivity in private data analysis. Theory of Cryptography: Third Theory of Cryptography Conference, TCC 2006, New York, NY, USA, March 4-7, 2006. Proceedings 3, 265\u2013284. Springer","DOI":"10.1007\/11681878_14"},{"key":"6930_CR36","unstructured":"Eldan, R., & Russinovich, M. (2023). Who\u2019s harry potter? approximate unlearning in llms. arXiv preprint arXiv:2310.02238"},{"key":"6930_CR37","doi-asserted-by":"crossref","unstructured":"Elesedy, H., Esperan\u00e7a, P.M., Oprea, S.V., & Ozay, M. (2024). Lora-guard: Parameter-efficient guardrail adaptation for content moderation of large language models. arXiv preprint arXiv:2407.02987","DOI":"10.18653\/v1\/2024.emnlp-main.656"},{"issue":"1","key":"6930_CR38","doi-asserted-by":"publisher","first-page":"10129","DOI":"10.1038\/s41598-024-59616-0","volume":"14","author":"A Esposito","year":"2024","unstructured":"Esposito, A., Desolda, G., & Lanzilotti, R. (2024). The fine line between automation and augmentation in website usability evaluation. Scientific Reports, 14(1), 10129.","journal-title":"Scientific Reports"},{"key":"6930_CR39","unstructured":"Ethayarajh, K., Xu, W., Muennighoff, N., Jurafsky, D., & Kiela, D. (2024). Model alignment as prospect theoretic optimization. In: Forty-first International Conference on Machine Learning"},{"issue":"10","key":"6930_CR40","doi-asserted-by":"publisher","first-page":"1839","DOI":"10.3390\/electronics13101839","volume":"13","author":"CS Eze","year":"2024","unstructured":"Eze, C. S., & Shamir, L. (2024). Analysis and prevention of ai-based phishing email attacks. Electronics, 13(10), 1839.","journal-title":"Electronics"},{"key":"6930_CR41","unstructured":"Fayard, M. (2023). What Is AI Monitoring and Why Is It Important. https:\/\/coralogix.com\/blog\/ai-monitoring\/"},{"key":"6930_CR42","doi-asserted-by":"crossref","unstructured":"Fournier, F., Limonad, L., & Skarbovsky, I. (2024). Towards a benchmark for causal business process reasoning with llms. arXiv preprint arXiv:2406.05506","DOI":"10.1007\/978-3-031-78666-2_18"},{"key":"6930_CR43","unstructured":"Frery, J. (2023). Towards Encrypted Large Language Models with FHE. https:\/\/huggingface.co\/blog\/encrypted-llm"},{"key":"6930_CR44","doi-asserted-by":"crossref","unstructured":"Fu, Y., Shayegan, E., Abdullah, M.M.A., Zaree, P., Abu-Ghazaleh, N., & Dong, Y. (2024). Vulnerabilities of large language models to adversarial attacks. In: Chiruzzo, L., Lee, H.-y., Ribeiro, L. (eds.) Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics (Volume 5: Tutorial Abstracts), pp. 8\u20139. Association for Computational Linguistics, Bangkok, Thailand. https:\/\/doi.org\/10.18653\/v1\/2024.acl-tutorials.5 . https:\/\/aclanthology.org\/2024.acl-tutorials.5\/","DOI":"10.18653\/v1\/2024.acl-tutorials.5"},{"key":"6930_CR45","doi-asserted-by":"crossref","unstructured":"Gallegos, I.O., Rossi, R.A., Barrow, J., Tanjim, M.M., Kim, S., Dernoncourt, F., Yu, T., Zhang, R., & Ahmed, N.K. (2024). Bias and fairness in large language models: A survey. Computational Linguistics, 1\u201379","DOI":"10.1162\/coli_a_00524"},{"key":"6930_CR46","unstructured":"Ganguli, D., Lovitt, L., Kernion, J., Askell, A., Bai, Y., Kadavath, S., Mann, B., Perez, E., Schiefer, N., Ndousse, K., et al. (2022). Red teaming language models to reduce harms: Methods, scaling behaviors, and lessons learned. arXiv preprint arXiv:2209.07858"},{"key":"6930_CR47","doi-asserted-by":"publisher","unstructured":"Ge, S., Zhou, C., Hou, R., Khabsa, M., Wang, Y.-C., Wang, Q., Han, J., & Mao, Y. (2024). MART: Improving LLM safety with multi-round automatic red-teaming. In: Proceedings of the 2024 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies (Volume 1: Long Papers), pp. 1927\u20131937. Association for Computational Linguistics, Mexico City, Mexico. https:\/\/doi.org\/10.18653\/v1\/2024.naacl-long.107","DOI":"10.18653\/v1\/2024.naacl-long.107"},{"key":"6930_CR48","doi-asserted-by":"publisher","unstructured":"Gehman, S., Gururangan, S., Sap, M., Choi, Y., & Smith, N.A. (2020). RealToxicityPrompts: Evaluating neural toxic degeneration in language models. In: Findings of the Association for Computational Linguistics: EMNLP 2020, pp. 3356\u20133369. Association for Computational Linguistics, Online. https:\/\/doi.org\/10.18653\/v1\/2020.findings-emnlp.301","DOI":"10.18653\/v1\/2020.findings-emnlp.301"},{"key":"6930_CR49","unstructured":"G\u00e9mes, K., & Recski, G. (2021). Tuw-inf at germeval2021: Rule-based and hybrid methods for detecting toxic, engaging, and fact-claiming comments. Proceedings of the GermEval 2021 Shared Task on the Identification of Toxic, Engaging, and Fact-Claiming Comments, 69\u201375"},{"key":"6930_CR50","unstructured":"Ghosh, S., Varshney, P., Galinkin, E., & Parisien, C. (2024). Aegis: Online adaptive ai content safety moderation with ensemble of llm experts. arXiv preprint arXiv:2404.05993"},{"key":"6930_CR51","doi-asserted-by":"crossref","unstructured":"Gillespie, N., Lockey, S., Curtis, C., Pool, J., & Akbari, A. (2023). Trust in artificial intelligence: A global study. The University of Queensland and KPMG Australia 10","DOI":"10.14264\/00d3c94"},{"key":"6930_CR52","unstructured":"Ginart, A., Maaten, L., Zou, J., & Guo, C. (2022). Submix: Practical private prediction for large-scale language models. arXiv preprint arXiv:2201.00971"},{"issue":"110","key":"6930_CR53","first-page":"1","volume":"78","author":"O Goldreich","year":"1998","unstructured":"Goldreich, O. (1998). Secure multi-party computation. Manuscript. Preliminary version, 78(110), 1\u2013108.","journal-title":"Secure multi-party computation. Manuscript. Preliminary version"},{"key":"6930_CR54","doi-asserted-by":"publisher","unstructured":"G\u00f3mez-Rodr\u00edguez, C., & Williams, P. (2023). A confederacy of models: a comprehensive evaluation of LLMs on creative writing. In: Findings of the Association for Computational Linguistics: EMNLP 2023, pp. 14504\u201314528. Association for Computational Linguistics, Singapore. https:\/\/doi.org\/10.18653\/v1\/2023.findings-emnlp.966","DOI":"10.18653\/v1\/2023.findings-emnlp.966"},{"key":"6930_CR55","unstructured":"Google: Perspective API (2017). https:\/\/perspectiveapi.com\/"},{"key":"6930_CR56","unstructured":"Goyal, T., Li, J.J., & Durrett, G. (2022). News summarization and evaluation in the era of gpt-3. arXiv preprint arXiv:2209.12356"},{"key":"6930_CR57","doi-asserted-by":"crossref","unstructured":"Greshake, K., Abdelnabi, S., Mishra, S., Endres, C., Holz, T., & Fritz, M. (2023). Not what you\u2019ve signed up for: Compromising real-world llm-integrated applications with indirect prompt injection. In: Proceedings of the 16th ACM Workshop on Artificial Intelligence and Security, pp. 79\u201390","DOI":"10.1145\/3605764.3623985"},{"key":"6930_CR58","doi-asserted-by":"crossref","unstructured":"Gupta, M., Akiri, C., Aryal, K., Parker, E., & Praharaj, L. (2023). From chatgpt to threatgpt: Impact of generative ai in cybersecurity and privacy. IEEE Access","DOI":"10.1109\/ACCESS.2023.3300381"},{"key":"6930_CR59","unstructured":"He, J., Jiang, W., Hou, G., Fan, W., Zhang, R., & Li, H. (2024). Talk too much: Poisoning large language models under token limit. arXiv preprint arXiv:2404.14795"},{"key":"6930_CR60","doi-asserted-by":"crossref","unstructured":"He, P., Xu, H., Xing, Y., Liu, H., Yamada, M., & Tang, J. (2024). Data poisoning for in-context learning. arXiv preprint arXiv:2402.02160","DOI":"10.18653\/v1\/2025.findings-naacl.91"},{"key":"6930_CR61","unstructured":"Hong, Z.-W., Shenfeld, I., Wang, T.-H., Chuang, Y.-S., Pareja, A., Glass, J.R., Srivastava, A., & Agrawal, P. (2024). Curiosity-driven red-teaming for large language models. In: The Twelfth International Conference on Learning Representations"},{"key":"6930_CR62","doi-asserted-by":"crossref","unstructured":"Hou, A., Zhang, J., He, T., Wang, Y., Chuang, Y.-S., Wang, H., Shen, L., Van\u00a0Durme, B., Khashabi, D., & Tsvetkov, Y. (2024). Semstamp: A semantic watermark with paraphrastic robustness for text generation. In: Proceedings of the 2024 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies (Volume 1: Long Papers), 4067\u20134082","DOI":"10.18653\/v1\/2024.naacl-long.226"},{"key":"6930_CR63","doi-asserted-by":"crossref","unstructured":"Hou, X., Zhao, Y., & Wang, H. (2024). On the (in) security of llm app stores. arXiv preprint arXiv:2407.08422","DOI":"10.1109\/SP61157.2025.00117"},{"key":"6930_CR64","doi-asserted-by":"publisher","unstructured":"Huang, J., Shao, H., & Chang, K.C.-C. (2022). Are large pre-trained language models leaking your personal information? In: Findings of the Association for Computational Linguistics: EMNLP 2022, pp. 2038\u20132047. Association for Computational Linguistics, Abu Dhabi, United Arab Emirates. https:\/\/doi.org\/10.18653\/v1\/2022.findings-emnlp.148","DOI":"10.18653\/v1\/2022.findings-emnlp.148"},{"key":"6930_CR65","first-page":"2198","volume":"2023","author":"Y Huang","year":"2023","unstructured":"Huang, Y., Zhuo, T. Y., Xu, Q., Hu, H., Yuan, X., & Chen, C. (2023). Training-free lexical backdoor attacks on language models. Proceedings of the ACM Web Conference, 2023, 2198\u20132208.","journal-title":"Proceedings of the ACM Web Conference"},{"key":"6930_CR66","doi-asserted-by":"publisher","first-page":"126265","DOI":"10.52202\/079017-4011","volume":"37","author":"X Hu","year":"2024","unstructured":"Hu, X., Chen, P.-Y., & Ho, T.-Y. (2024). Gradient cuff: Detecting jailbreak attacks on large language models by exploring refusal loss landscapes. Advances in Neural Information Processing Systems, 37, 126265\u2013126296.","journal-title":"Advances in Neural Information Processing Systems"},{"issue":"162","key":"6930_CR67","first-page":"1","volume":"800","author":"VC Hu","year":"2013","unstructured":"Hu, V. C., Ferraiolo, D., Kuhn, R., Friedman, A. R., Lang, A. J., Cogdell, M. M., Schnitzer, A., Sandlin, K., Miller, R., Scarfone, K., et al. (2013). Guide to attribute based access control (abac) definition and considerations (draft). NIST special publication, 800(162), 1\u201354.","journal-title":"NIST special publication"},{"key":"6930_CR68","doi-asserted-by":"crossref","unstructured":"Hui, B., Yuan, H., Gong, N., Burlina, P., & Cao, Y. (2024). Pleak: Prompt leaking attacks against large language model applications. arXiv preprint arXiv:2405.06823","DOI":"10.1145\/3658644.3670370"},{"key":"6930_CR69","doi-asserted-by":"crossref","unstructured":"Hung, K.-H., Ko, C.-Y., Rawat, A., Chung, I.-H., Hsu, W.H., & Chen, P.-Y. (2025). Attention tracker: Detecting prompt injection attacks in LLMs. In: Chiruzzo, L., Ritter, A., Wang, L. (eds.) Findings of the Association for Computational Linguistics: NAACL 2025, pp. 2309\u20132322. Association for Computational Linguistics, Albuquerque, New Mexico. https:\/\/doi.org\/10.18653\/v1\/2025.findings-naacl.123 . https:\/\/aclanthology.org\/2025.findings-naacl.123\/","DOI":"10.18653\/v1\/2025.findings-naacl.123"},{"key":"6930_CR70","doi-asserted-by":"publisher","unstructured":"Hussain, A., Rabin, M.R.I., & Alipour, M.A. (2024). Measuring impacts of poisoning on model parameters and embeddings for large language models of code. In: Proceedings of the 1st ACM International Conference on AI-Powered Software. AIware 2024, pp. 59\u201364. Association for Computing Machinery, New York, NY, USA. https:\/\/doi.org\/10.1145\/3664646.3664764","DOI":"10.1145\/3664646.3664764"},{"key":"6930_CR71","unstructured":"Inan, H., Upasani, K., Chi, J., Rungta, R., Iyer, K., Mao, Y., Tontchev, M., Hu, Q., Fuller, B., Testuggine, D., et al. (2023). Llama guard: Llm-based input-output safeguard for human-ai conversations. arXiv preprint arXiv:2312.06674"},{"key":"6930_CR72","unstructured":"Jagannatha, A., Rawat, B.P.S., & Yu, H. (2021). Membership inference attack susceptibility of clinical language models. arXiv preprint arXiv:2104.08305"},{"key":"6930_CR73","unstructured":"Jain, N., Schwarzschild, A., Wen, Y., Somepalli, G., Kirchenbauer, J., Chiang, P.-y., Goldblum, M., Saha, A., Geiping, J., & Goldstein, T. (2023). Baseline defenses for adversarial attacks against aligned language models. arXiv preprint arXiv:2309.00614"},{"key":"6930_CR74","unstructured":"Jiang, B., Jing, Y., Wu, T., Shen, T., Xiong, D., & Yang, Q. (2025). Automated progressive red teaming. In: Proceedings of the 31st International Conference on Computational Linguistics, pp. 3850\u20133864. Association for Computational Linguistics."},{"key":"6930_CR75","unstructured":"Jiang, S., Kadhe, S., Zhou, Y., Cai, L., & Baracaldo, N. (2024). Forcing generative models to degenerate ones: The power of data poisoning attacks. NeurIPS 2023 Workshop on Backdoors in Deep Learning - The Good, the Bad, and the Ugly"},{"key":"6930_CR76","doi-asserted-by":"publisher","unstructured":"Jiang, F., Xu, Z., Niu, L., Xiang, Z., Ramasubramanian, B., Li, B., & Poovendran, R. (2024). ArtPrompt: ASCII art-based jailbreak attacks against aligned LLMs. In: Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), pp. 15157\u201315173. Association for Computational Linguistics, Bangkok, Thailand. https:\/\/doi.org\/10.18653\/v1\/2024.acl-long.809","DOI":"10.18653\/v1\/2024.acl-long.809"},{"key":"6930_CR77","unstructured":"Kandpal, N., Wallace, E., & Raffel, C. (2022). Deduplicating training data mitigates privacy risks in language models. In: International Conference on Machine Learning, 10697\u201310707. PMLR"},{"key":"6930_CR78","doi-asserted-by":"crossref","unstructured":"Kang, D., Li, X., Stoica, I., Guestrin, C., Zaharia, M., & Hashimoto, T. (2024). Exploiting programmatic behavior of llms: Dual-use through standard security attacks. In: 2024 IEEE Security and Privacy Workshops (SPW), 132\u2013143. IEEE","DOI":"10.1109\/SPW63631.2024.00018"},{"key":"6930_CR79","doi-asserted-by":"publisher","DOI":"10.1016\/j.techsoc.2023.102232","volume":"73","author":"HO Khogali","year":"2023","unstructured":"Khogali, H. O., & Mekid, S. (2023). The blended future of automation and ai: Examining some long-term societal and ethical impact features. Technology in Society, 73, Article 102232.","journal-title":"Technology in Society"},{"key":"6930_CR80","unstructured":"Kirchenbauer, J., Geiping, J., Wen, Y., Katz, J., Miers, I., & Goldstein, T. (2023). A watermark for large language models. In: International Conference on Machine Learning, 17061\u201317084. PMLR"},{"key":"6930_CR81","doi-asserted-by":"publisher","unstructured":"Kong, A., Zhao, S., Chen, H., Li, Q., Qin, Y., Sun, R., Zhou, X., Wang, E., & Dong, X. (2024). Better zero-shot reasoning with role-play prompting. In: Proceedings of the 2024 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies (Volume 1: Long Papers), pp. 4099\u20134113. Association for Computational Linguistics, Mexico City, Mexico. https:\/\/doi.org\/10.18653\/v1\/2024.naacl-long.228","DOI":"10.18653\/v1\/2024.naacl-long.228"},{"key":"6930_CR82","unstructured":"Kour, G., Zalmanovici, M., Zwerdling, N., Goldbraich, E., Fandina, O., Anaby\u00a0Tavor, A., Raz, O., & Farchi, E. (2023). Unveiling safety vulnerabilities of large language models. In: Proceedings of the Third Workshop on Natural Language Generation, Evaluation, and Metrics (GEM), pp. 111\u2013127. Association for Computational Linguistics, Singapore"},{"issue":"1","key":"6930_CR83","doi-asserted-by":"publisher","first-page":"104","DOI":"10.1017\/XPS.2020.37","volume":"9","author":"S Kreps","year":"2022","unstructured":"Kreps, S., McCain, R. M., & Brundage, M. (2022). All the news that\u2019s fit to fabricate: Ai-generated text as a tool of media misinformation. Journal of Experimental Political Science, 9(1), 104\u2013117.","journal-title":"Journal of Experimental Political Science"},{"key":"6930_CR84","doi-asserted-by":"publisher","unstructured":"Kurita, K., Michel, P., & Neubig, G. (2020). Weight poisoning attacks on pretrained models. In: Proceedings of the 58th Annual Meeting of the Association for Computational Linguistics, pp. 2793\u20132806. Association for Computational Linguistics, Online. https:\/\/doi.org\/10.18653\/v1\/2020.acl-main.249","DOI":"10.18653\/v1\/2020.acl-main.249"},{"key":"6930_CR85","doi-asserted-by":"publisher","unstructured":"Lee, K., Ippolito, D., Nystrom, A., Zhang, C., Eck, D., Callison-Burch, C., & Carlini, N. (2022). Deduplicating training data makes language models better. In: Proceedings of the 60th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), pp. 8424\u20138445. Association for Computational Linguistics, Dublin, Ireland. https:\/\/doi.org\/10.18653\/v1\/2022.acl-long.577","DOI":"10.18653\/v1\/2022.acl-long.577"},{"key":"6930_CR86","unstructured":"Lee, H., Phatale, S., Mansoor, H., Mesnard, T., Ferret, J., Lu, K.R., Bishop, C., Hall, E., Carbune, V., Rastogi, A., et al. (2024). Rlaif vs. rlhf: Scaling reinforcement learning from human feedback with ai feedback. In: International Conference on Machine Learning, pp. 26874\u201326901. PMLR"},{"key":"6930_CR87","doi-asserted-by":"crossref","unstructured":"Li, N., Gao, C., Li, M., Li, Y., & Liao, Q. (2024). EconAgent: Large language model-empowered agents for simulating macroeconomic activities. In: Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), pp. 15523\u201315536. Association for Computational Linguistics, Bangkok, Thailand","DOI":"10.18653\/v1\/2024.acl-long.829"},{"key":"6930_CR88","unstructured":"Li, D., Wang, H., Shao, R., Guo, H., Xing, E., & Zhang, H. (2023). MPCFORMER: FAST, PERFORMANT AND PRIVATE TRANSFORMER INFERENCE WITH MPC. In: The Eleventh International Conference on Learning Representations"},{"key":"6930_CR89","unstructured":"Li, S., Yao, L., Gao, J., Zhang, L., & Li, Y. (2024). Double-i watermark: Protecting model copyright for llm fine-tuning. arXiv preprint arXiv:2402.14883"},{"key":"6930_CR90","doi-asserted-by":"crossref","unstructured":"Li, X., Zhang, Y., Lou, R., Wu, C., & Wang, J. (2024). Chain-of-scrutiny: Detecting backdoor attacks for large language models. arXiv preprint arXiv:2406.05948","DOI":"10.18653\/v1\/2025.findings-acl.401"},{"key":"6930_CR91","unstructured":"Li, X., Zhou, Z., Zhu, J., Yao, J., Liu, T., & Han, B. (2023). Deepinception: Hypnotize large language model to be jailbreaker. arXiv preprint arXiv:2311.03191"},{"key":"6930_CR92","doi-asserted-by":"publisher","unstructured":"Lin, S., Hilton, J., & Evans, O. (2022). TruthfulQA: Measuring how models mimic human falsehoods. In: Proceedings of the 60th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), pp. 3214\u20133252. Association for Computational Linguistics, Dublin, Ireland. https:\/\/doi.org\/10.18653\/v1\/2022.acl-long.229","DOI":"10.18653\/v1\/2022.acl-long.229"},{"key":"6930_CR93","doi-asserted-by":"crossref","unstructured":"Liu, Y., Deng, G., Xu, Z., Li, Y., Zheng, Y., Zhang, Y., Zhao, L., Zhang, T., Wang, K., & Liu, Y. (2023). Jailbreaking chatgpt via prompt engineering: An empirical study. arXiv preprint arXiv:2305.13860","DOI":"10.1145\/3663530.3665021"},{"key":"6930_CR94","doi-asserted-by":"publisher","DOI":"10.1145\/3682068","author":"X-Y Liu","year":"2024","unstructured":"Liu, X.-Y., Zhu, R., Zha, D., Gao, J., Zhong, S., White, M., & Qiu, M. (2024). Differentially private low-rank adaptation of large language model using federated learning. ACM Transactions on Management Information Systems. https:\/\/doi.org\/10.1145\/3682068. Just Accepted.","journal-title":"ACM Transactions on Management Information Systems"},{"key":"6930_CR95","unstructured":"Low, Y.S., Jackson, M.L., Hyde, R.J., Brown, R.E., Sanghavi, N.M., Baldwin, J.D., Pike, C.W., Muralidharan, J., Hui, G., Alexander, N., et al. (2024). Answering real-world clinical questions using large language model based systems. arXiv preprint arXiv:2407.00541"},{"key":"6930_CR96","unstructured":"Lu, T., & Koehn, P. (2024). Every language counts: Learn and unlearn in multilingual llms. arXiv preprint arXiv:2406.13748"},{"key":"6930_CR97","doi-asserted-by":"crossref","unstructured":"Meng, Y., Xia, M., & Chen, D. (2024). Simpo: Simple preference optimization with a reference-free reward. arXiv preprint arXiv:2405.14734","DOI":"10.52202\/079017-3946"},{"key":"6930_CR98","doi-asserted-by":"publisher","first-page":"231","DOI":"10.1016\/j.eswa.2016.01.028","volume":"53","author":"M Moghimi","year":"2016","unstructured":"Moghimi, M., & Varjani, A. Y. (2016). New rule-based phishing detection method. Expert Systems with Applications, 53, 231\u2013242.","journal-title":"Expert Systems with Applications"},{"key":"6930_CR99","unstructured":"Mozes, M., He, X., Kleinberg, B., & Griffin, L.D. (2023). Use of llms for illicit purposes: Threats, prevention measures, and vulnerabilities. arXiv preprint arXiv:2308.12833"},{"key":"6930_CR100","unstructured":"Mu, N., Chen, S., Wang, Z., Chen, S., Karamardian, D., Aljeraisy, L., Hendrycks, D., & Wagner, D. (2023). Can llms follow simple rules? arXiv preprint arXiv:2311.04235"},{"key":"6930_CR101","doi-asserted-by":"crossref","unstructured":"Nagireddy, M., Chiazor, L., Singh, M., & Baldini, I. (2024) Socialstigmaqa: A benchmark to uncover stigma amplification in generative language models. Proceedings of the AAAI Conference on Artificial Intelligence, 38, 21454\u201321462","DOI":"10.1609\/aaai.v38i19.30142"},{"key":"6930_CR102","unstructured":"Nasr, M., Carlini, N., Hayase, J., Jagielski, M., Cooper, A.F., Ippolito, D., Choquette-Choo, C.A., Wallace, E., Tram\u00e8r, F., & Lee, K. (2023). Scalable extraction of training data from (production) language models. arXiv preprint arXiv:2311.17035"},{"key":"6930_CR103","doi-asserted-by":"crossref","unstructured":"Nghiem, H., Prindle, J., Zhao, J., & Daum\u00e9\u00a0III, H. (2024). \u201c you gotta be a doctor, lin\u201d: An investigation of name-based bias of large language models in employment recommendations. arXiv preprint arXiv:2406.12232","DOI":"10.18653\/v1\/2024.emnlp-main.413"},{"key":"6930_CR104","doi-asserted-by":"publisher","unstructured":"Oliynyk, D., Mayer, R., & Rauber, A. (2023). I know what you trained last summer: A survey on stealing machine learning models and defences. ACM Comput. Surv. 55(14s) https:\/\/doi.org\/10.1145\/3595292","DOI":"10.1145\/3595292"},{"key":"6930_CR105","doi-asserted-by":"publisher","DOI":"10.1016\/j.techfore.2022.121763","volume":"181","author":"N Omrani","year":"2022","unstructured":"Omrani, N., Rivieccio, G., Fiore, U., Schiavone, F., & Agreda, S. G. (2022). To trust or not to trust? an assessment of trust in ai-based systems: Concerns, ethics and contexts. Technological Forecasting and Social Change, 181, Article 121763.","journal-title":"Technological Forecasting and Social Change"},{"key":"6930_CR106","unstructured":"OpenAI: (2024) Introducing OpenAI o1-preview. https:\/\/openai.com\/index\/introducing-openai-o1-preview\/"},{"key":"6930_CR107","unstructured":"Ophir\u00a0Dror, B.L. (2024). Riding the RAG Trail: Access, Permissions and Context. https:\/\/www.lasso.security\/blog\/riding-the-rag-trail-access-permissions-and-context"},{"key":"6930_CR108","doi-asserted-by":"publisher","first-page":"27730","DOI":"10.52202\/068431-2011","volume":"35","author":"L Ouyang","year":"2022","unstructured":"Ouyang, L., Wu, J., Jiang, X., Almeida, D., Wainwright, C., Mishkin, P., Zhang, C., Agarwal, S., Slama, K., Ray, A., et al. (2022). Training language models to follow instructions with human feedback. Advances in Neural Information Processing Systems, 35, 27730\u201327744.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"6930_CR109","unstructured":"Pal, A., Karkhanis, D., Dooley, S., Roberts, M., Naidu, S., & White, C. (2024). Smaug: Fixing failure modes of preference optimisation with dpo-positive. arXiv preprint arXiv:2402.13228"},{"key":"6930_CR110","doi-asserted-by":"crossref","unstructured":"Pang, K., Qi, T., Wu, C., & Bai, M. (2024). Adaptive and robust watermark against model extraction attack. arXiv preprint arXiv:2405.02365","DOI":"10.1109\/TIFS.2025.3530691"},{"key":"6930_CR111","doi-asserted-by":"crossref","unstructured":"Papernot, N., McDaniel, P., Goodfellow, I., Jha, S., Celik, Z.B., & Swami, A. (2017). Practical black-box attacks against machine learning. Proceedings of the 2017 ACM on Asia Conference on Computer and Communications Security, 506\u2013519","DOI":"10.1145\/3052973.3053009"},{"key":"6930_CR112","unstructured":"Pathmanathan, P., Chakraborty, S., Liu, X., Liang, Y., & Huang, F. (2024). Is poisoning a real threat to LLM alignment? maybe more so than you think. In: ICML 2024 Workshop on Models of Human Feedback for AI Alignment"},{"key":"6930_CR113","unstructured":"Peixoto, T.C., Canuto, O., & Jordan, L. (2024). Ai and the future of government: Unexpected effects and critical challenges. Policy Center For The New South"},{"key":"6930_CR114","doi-asserted-by":"crossref","unstructured":"Penedo, G., Malartic, Q., Hesslow, D., Cojocaru, R., Alobeidli, H., Cappelli, A., Pannier, B., Almazrouei, E., & Launay, J. (2024). The refinedweb dataset for falcon llm: outperforming curated corpora with web data only. In: Proceedings of the 37th International Conference on Neural Information Processing Systems. NIPS \u201923. Curran Associates Inc., Red Hook, NY, USA","DOI":"10.52202\/075280-3464"},{"key":"6930_CR115","doi-asserted-by":"publisher","unstructured":"Peng, W., Yi, J., Wu, F., Wu, S., Bin\u00a0Zhu, B., Lyu, L., Jiao, B., Xu, T., Sun, G., & Xie, X. (2023). Are you copying my model? protecting the copyright of large language models for EaaS via backdoor watermark. In: Proceedings of the 61st Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), pp. 7653\u20137668. Association for Computational Linguistics, Toronto, Canada. https:\/\/doi.org\/10.18653\/v1\/2023.acl-long.423","DOI":"10.18653\/v1\/2023.acl-long.423"},{"key":"6930_CR116","unstructured":"Perez, F., & Ribeiro, I. (2022). Ignore previous prompt: Attack techniques for language models. In: NeurIPS ML Safety Workshop"},{"key":"6930_CR117","doi-asserted-by":"publisher","unstructured":"Perez, E., Huang, S., Song, F., Cai, T., Ring, R., Aslanides, J., Glaese, A., McAleese, N., & Irving, G. (2022). Red teaming language models with language models. In: Proceedings of the 2022 Conference on Empirical Methods in Natural Language Processing, pp. 3419\u20133448. Association for Computational Linguistics, Abu Dhabi, United Arab Emirates. https:\/\/doi.org\/10.18653\/v1\/2022.emnlp-main.225","DOI":"10.18653\/v1\/2022.emnlp-main.225"},{"key":"6930_CR118","doi-asserted-by":"crossref","unstructured":"Plaza-del-Arco, F., Curry, A., Cercas\u00a0Curry, A., Abercrombie, G., & Hovy, D. (2024). Angry men, sad women: Large language models reflect gendered stereotypes in emotion attribution. In: Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), pp. 7682\u20137696. Association for Computational Linguistics, Bangkok, Thailand","DOI":"10.18653\/v1\/2024.acl-long.415"},{"key":"6930_CR119","doi-asserted-by":"publisher","unstructured":"Qi, F., Li, M., Chen, Y., Zhang, Z., Liu, Z., Wang, Y., & Sun, M. (2021). Hidden killer: Invisible textual backdoor attacks with syntactic trigger. In: Proceedings of the 59th Annual Meeting of the Association for Computational Linguistics and the 11th International Joint Conference on Natural Language Processing (Volume 1: Long Papers), pp. 443\u2013453. Association for Computational Linguistics, Online. https:\/\/doi.org\/10.18653\/v1\/2021.acl-long.37","DOI":"10.18653\/v1\/2021.acl-long.37"},{"key":"6930_CR120","unstructured":"Qi, X., Zeng, Y., Xie, T., Chen, P.-Y., Jia, R., Mittal, P., & Henderson, P. (2024). Fine-tuning aligned language models compromises safety, even when users do not intend to! In: ICLR"},{"key":"6930_CR121","unstructured":"Qiang, Y., Zhou, X., Zade, S.Z., Roshani, M.A., Zytko, D., & Zhu, D. (2024). Learning to poison large language models during instruction tuning. arXiv preprint arXiv:2402.13459"},{"issue":"8","key":"6930_CR122","first-page":"9","volume":"1","author":"A Radford","year":"2019","unstructured":"Radford, A., Wu, J., Child, R., Luan, D., Amodei, D., Sutskever, I., et al. (2019). Language models are unsupervised multitask learners. OpenAI blog, 1(8), 9.","journal-title":"OpenAI blog"},{"key":"6930_CR123","doi-asserted-by":"crossref","unstructured":"Rafailov, R., Sharma, A., Mitchell, E., Manning, C.D., Ermon, S., & Finn, C. (2024). Direct preference optimization: Your language model is secretly a reward model. Advances in Neural Information Processing Systems 36","DOI":"10.52202\/075280-2338"},{"key":"6930_CR124","unstructured":"Rahman, N., & Santacana, E. (2023). Beyond fair use: Legal risk evaluation for training llms on copyrighted text. In: ICML Workshop on Generative AI and Law"},{"key":"6930_CR125","unstructured":"Rando, J., & Tram\u00e8r, F. (2024). Universal jailbreak backdoors from poisoned human feedback. In: The Twelfth International Conference on Learning Representations"},{"key":"6930_CR126","unstructured":"Robey, A., Wong, E., Hassani, H., & Pappas, G.J. (2023). Smoothllm: Defending large language models against jailbreaking attacks. arXiv preprint arXiv:2310.03684"},{"key":"6930_CR127","unstructured":"Roziere, B., Gehring, J., Gloeckle, F., Sootla, S., Gat, I., Tan, X.E., Adi, Y., Liu, J., Sauvestre, R., Remez, T., et al. (2023). Code llama: Open foundation models for code. arXiv preprint arXiv:2308.12950"},{"key":"6930_CR128","doi-asserted-by":"crossref","unstructured":"Sandhu, R.S. (1998). Role-based access control. In: Advances in Computers vol. 46, pp. 237\u2013286. Elsevier.","DOI":"10.1016\/S0065-2458(08)60206-5"},{"key":"6930_CR129","unstructured":"Sani, L., Iacob, A., Cao, Z., Marino, B., Gao, Y., Paulik, T., Zhao, W., Shen, W.F., Aleksandrov, P., Qiu, X., et al. (2024). The future of large language model pre-training is federated. arXiv preprint arXiv:2405.10853"},{"key":"6930_CR130","doi-asserted-by":"crossref","unstructured":"Schulhoff, S., Pinto, J., Khan, A., Bouchard, L.-F., Si, C., Anati, S., Tagliabue, V., Kost, A., Carnahan, C., & Boyd-Graber, J. (2023). Ignore this title and hackaprompt: Exposing systemic vulnerabilities of llms through a global prompt hacking competition. In: Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing, 4945\u20134977","DOI":"10.18653\/v1\/2023.emnlp-main.302"},{"key":"6930_CR131","unstructured":"Sha, Z., & Zhang, Y. (2024). Prompt stealing attacks against large language models. arXiv preprint arXiv:2402.12959"},{"key":"6930_CR132","doi-asserted-by":"crossref","unstructured":"Shao, H., Huang, J., Zheng, S., & Chang, K. (2024). Quantifying association capabilities of large language models and its implications on privacy leakage. In: Findings of the Association for Computational Linguistics: EACL 2024, 814\u2013825","DOI":"10.18653\/v1\/2024.findings-eacl.54"},{"key":"6930_CR133","doi-asserted-by":"publisher","unstructured":"Shi, W., Shea, R., Chen, S., Zhang, C., Jia, R., & Yu, Z. (2022). Just fine-tune twice: Selective differential privacy for large language models. In: Proceedings of the 2022 Conference on Empirical Methods in Natural Language Processing, pp. 6327\u20136340. Association for Computational Linguistics, Abu Dhabi, United Arab Emirates. https:\/\/doi.org\/10.18653\/v1\/2022.emnlp-main.425","DOI":"10.18653\/v1\/2022.emnlp-main.425"},{"key":"6930_CR134","unstructured":"Shumailov, I., Hayes, J., Triantafillou, E., Ortiz-Jimenez, G., Papernot, N., Jagielski, M., Yona, I., Howard, H., & Bagdasaryan, E. (2024). Ununlearning: Unlearning is not sufficient for content regulation in advanced generative ai. arXiv preprint arXiv:2407.00106"},{"key":"6930_CR135","doi-asserted-by":"publisher","first-page":"61836","DOI":"10.52202\/075280-2703","volume":"36","author":"M Shu","year":"2023","unstructured":"Shu, M., Wang, J., Zhu, C., Geiping, J., Xiao, C., & Goldstein, T. (2023). On the exploitability of instruction tuning. Advances in Neural Information Processing Systems, 36, 61836\u201361856.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"6930_CR136","unstructured":"Si, W.M., Backes, M., & Zhang, Y. (2023). Mondrian: Prompt abstraction attack against large language models for cheaper api pricing. arXiv preprint arXiv:2308.03558"},{"key":"6930_CR137","doi-asserted-by":"crossref","unstructured":"Song, F., Yu, B., Li, M., Yu, H., Huang, F., Li, Y., & Wang, H. (2024). Preference ranking optimization for human alignment. Proceedings of the AAAI Conference on Artificial Intelligence, 38, 18990\u201318998","DOI":"10.1609\/aaai.v38i17.29865"},{"key":"6930_CR138","unstructured":"Srivastava, S., Mardziel, P., Zhang, Z., Ahlawat, A., Datta, A., & Mitchell, J.C. (2024). De-amplifying bias from differential privacy in language model fine-tuning. arXiv preprint arXiv:2402.04489"},{"key":"6930_CR139","unstructured":"Staab, R., Vero, M., Balunovic, M., & Vechev, M. (2024). Beyond memorization: Violating privacy via inference with large language models. In: The Twelfth International Conference on Learning Representations"},{"key":"6930_CR140","unstructured":"Staab, R., Vero, M., Balunovic, M., & Vechev, M. (2025). Language models are advanced anonymizers. In: The Thirteenth International Conference on Learning Representations"},{"key":"6930_CR141","doi-asserted-by":"crossref","unstructured":"Tamber, M.S., Xian, J., & Lin, J. (2024). Can\u2019t hide behind the api: Stealing black-box commercial embedding models. arXiv preprint arXiv:2406.09355","DOI":"10.18653\/v1\/2025.findings-naacl.104"},{"issue":"8","key":"6930_CR142","doi-asserted-by":"publisher","first-page":"1930","DOI":"10.1038\/s41591-023-02448-8","volume":"29","author":"AJ Thirunavukarasu","year":"2023","unstructured":"Thirunavukarasu, A. J., Ting, D. S. J., Elangovan, K., Gutierrez, L., Tan, T. F., & Ting, D. S. W. (2023). Large language models in medicine. Nature Medicine, 29(8), 1930\u20131940.","journal-title":"Nature Medicine"},{"key":"6930_CR143","unstructured":"Touvron, H., Martin, L., Stone, K., Albert, P., Almahairi, A., Babaei, Y., Bashlykov, N., Batra, S., Bhargava, P., Bhosale, S., et al. (2023). Llama 2: Open foundation and fine-tuned chat models. arXiv preprint arXiv:2307.09288"},{"key":"6930_CR144","unstructured":"Toyer, S., Watkins, O., Mendes, E.A., Svegliato, J., Bailey, L., Wang, T., Ong, I., Elmaaroufi, K., Abbeel, P., Darrell, T., Ritter, A., & Russell, S. (2024). Tensor trust: Interpretable prompt injection attacks from an online game. In: The Twelfth International Conference on Learning Representations"},{"key":"6930_CR145","unstructured":"Wan, A., Wallace, E., Shen, S., & Klein, D. (2023). Poisoning language models during instruction tuning. International Conference on Machine Learning, 35413\u201335425. PMLR"},{"key":"6930_CR146","doi-asserted-by":"crossref","unstructured":"Wang, B., Chen, W., Pei, H., Xie, C., Kang, M., Zhang, C., Xu, C., Xiong, Z., Dutta, R., Schaeffer, R., et al. (2024). Decodingtrust: A comprehensive assessment of trustworthiness in gpt models. Advances in Neural Information Processing Systems 36","DOI":"10.52202\/075280-1361"},{"key":"6930_CR147","doi-asserted-by":"publisher","unstructured":"Wang, Y., Kordi, Y., Mishra, S., Liu, A., Smith, N.A., Khashabi, D., & Hajishirzi, H. (2023). Self-instruct: Aligning language models with self-generated instructions. In: Proceedings of the 61st Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), 13484\u201313508. Association for Computational Linguistics, Toronto, Canada. https:\/\/doi.org\/10.18653\/v1\/2023.acl-long.754","DOI":"10.18653\/v1\/2023.acl-long.754"},{"key":"6930_CR148","doi-asserted-by":"publisher","unstructured":"Wang, Z., Ma, J., Wang, X., Hu, J., Qin, Z., & Ren, K. (2022). Threats to training: A survey of poisoning attacks and defenses on machine learning systems. ACM Comput. Surv. 55(7) https:\/\/doi.org\/10.1145\/3538707","DOI":"10.1145\/3538707"},{"key":"6930_CR149","doi-asserted-by":"publisher","unstructured":"Wang, N., Peng, Z.y., Que, H., Liu, J., Zhou, W., Wu, Y., Guo, H., Gan, R., Ni, Z., Yang, J., Zhang, M., Zhang, Z., Ouyang, W., Xu, K., Huang, W., Fu, J., & Peng, J. (2024). RoleLLM: Benchmarking, eliciting, and enhancing role-playing abilities of large language models. In: Findings of the Association for Computational Linguistics ACL 2024, 14743\u201314777. Association for Computational Linguistics, Bangkok, Thailand and virtual meeting. https:\/\/doi.org\/10.18653\/v1\/2024.findings-acl.878","DOI":"10.18653\/v1\/2024.findings-acl.878"},{"key":"6930_CR150","unstructured":"Wang, X., Wang, W., Ji, Z., Li, Z., Ma, P., Wu, D., & Wang, S. (2025). Stshield: Single-token sentinel for real-time jailbreak detection in large language models. arXiv preprint arXiv:2503.17932"},{"key":"6930_CR151","unstructured":"Wang, X., Wang, Z., Liu, J., Chen, Y., Yuan, L., Peng, H., & Ji, H. (2024). MINT: Evaluating LLMs in multi-turn interaction with tools and language feedback. The Twelfth International Conference on Learning Representations"},{"key":"6930_CR152","doi-asserted-by":"publisher","unstructured":"Wang, J., Wu, J., Chen, M., Vorobeychik, Y., & Xiao, C. (2024). RLHFPoison: Reward poisoning attack for reinforcement learning with human feedback in large language models. In: Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), 2551\u20132570. Association for Computational Linguistics, Bangkok, Thailand. https:\/\/doi.org\/10.18653\/v1\/2024.acl-long.140","DOI":"10.18653\/v1\/2024.acl-long.140"},{"key":"6930_CR153","doi-asserted-by":"crossref","unstructured":"Wei, A., Haghtalab, N., & Steinhardt, J. (2024). Jailbroken: How does llm safety training fail? Advances in Neural Information Processing Systems 36","DOI":"10.52202\/075280-3508"},{"key":"6930_CR154","unstructured":"Wei, Z., Wang, Y., & Wang, Y. (2023). Jailbreak and guard aligned language models with only few in-context demonstrations. arXiv preprint arXiv:2310.06387"},{"key":"6930_CR155","doi-asserted-by":"crossref","unstructured":"Wu, J., Guo, J., & Hooi, B. (2024). Fake news in sheep\u2019s clothing: Robust fake news detection against llm-empowered style attacks. Proceedings of the 30th ACM SIGKDD Conference on Knowledge Discovery and Data Mining, 3367\u20133378","DOI":"10.1145\/3637528.3671977"},{"key":"6930_CR156","unstructured":"Wu, T., Panda, A., Wang, J.T., & Mittal, P. (2024). Privacy-preserving in-context learning for large language models. In: The Twelfth International Conference on Learning Representations"},{"key":"6930_CR157","unstructured":"Wu, J., Yang, S., Zhan, R., Yuan, Y., Wong, D.F., & Chao, L.S. (2023). A survey on llm-gernerated text detection: Necessity, methods, and future directions. arXiv preprint arXiv:2310.14724"},{"key":"6930_CR158","unstructured":"Xiang, Z., Jiang, F., Xiong, Z., Ramasubramanian, B., Poovendran, R., & Li, B. (2024). Badchain: Backdoor chain-of-thought prompting for large language models. In: NeurIPS 2023 Workshop on Backdoors in Deep Learning - The Good, the Bad, and the Ugly"},{"issue":"1","key":"6930_CR159","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1057\/s41599-024-03307-8","volume":"11","author":"A Xiao","year":"2024","unstructured":"Xiao, A., Xu, Z., Skare, M., Qin, Y., & Wang, X. (2024). Bridging the digital divide: the impact of technological innovation on income inequality and human interactions. Humanities and Social Sciences Communications, 11(1), 1\u201318.","journal-title":"Humanities and Social Sciences Communications"},{"key":"6930_CR160","doi-asserted-by":"crossref","unstructured":"Xie, Y., Fang, M., Pi, R., & Gong, N. (2024). GradSafe: Detecting jailbreak prompts for LLMs via safety-critical gradient analysis. In: Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), pp. 507\u2013518. Association for Computational Linguistics, Bangkok, Thailand","DOI":"10.18653\/v1\/2024.acl-long.30"},{"key":"6930_CR161","doi-asserted-by":"publisher","unstructured":"Xu, C., Guo, D., Duan, N., & McAuley, J. (2023). Baize: An open-source chat model with parameter-efficient tuning on self-chat data. In: Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing, pp. 6268\u20136278. Association for Computational Linguistics, Singapore. https:\/\/doi.org\/10.18653\/v1\/2023.emnlp-main.385","DOI":"10.18653\/v1\/2023.emnlp-main.385"},{"key":"6930_CR162","unstructured":"Xu, Z., Jiang, F., Niu, L., Deng, Y., Poovendran, R., Choi, Y., & Lin, B.Y. (2025). Magpie: Alignment data synthesis from scratch by prompting aligned LLMs with nothing. In: The Thirteenth International Conference on Learning Representations"},{"key":"6930_CR163","doi-asserted-by":"crossref","unstructured":"Xu, J., Ju, D., Li, M., Boureau, Y.-L., Weston, J., & Dinan, E. (2021). Bot-adversarial dialogue for safe conversational agents. In: Proceedings of the 2021 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, 2950\u20132968","DOI":"10.18653\/v1\/2021.naacl-main.235"},{"key":"6930_CR164","doi-asserted-by":"publisher","unstructured":"Xu, J., Ma, M., Wang, F., Xiao, C., & Chen, M. (2024). Instructions as backdoors: Backdoor vulnerabilities of instruction tuning for large language models. In: Proceedings of the 2024 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies (Volume 1: Long Papers), pp. 3111\u20133126. Association for Computational Linguistics, Mexico City, Mexico. https:\/\/doi.org\/10.18653\/v1\/2024.naacl-long.171","DOI":"10.18653\/v1\/2024.naacl-long.171"},{"key":"6930_CR165","unstructured":"Xu, X., Yao, Y., & Liu, Y. (2024). Learning to watermark llm-generated text via reinforcement learning. arXiv preprint arXiv:2403.10553"},{"key":"6930_CR166","unstructured":"Xue, J., Zheng, M., Hu, Y., Liu, F., Chen, X., & Lou, Q. (2024). Badrag: Identifying vulnerabilities in retrieval augmented generation of large language models. arXiv preprint arXiv:2406.00083"},{"key":"6930_CR167","doi-asserted-by":"publisher","unstructured":"Yan, J., Gupta, V., & Ren, X. (2023). BITE: Textual backdoor attacks with iterative trigger injection. Proceedings of the 61st Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), 12951\u201312968. Association for Computational Linguistics, Toronto, Canada. https:\/\/doi.org\/10.18653\/v1\/2023.acl-long.725","DOI":"10.18653\/v1\/2023.acl-long.725"},{"key":"6930_CR168","doi-asserted-by":"crossref","unstructured":"Yan, B., Li, K., Xu, M., Dong, Y., Zhang, Y., Ren, Z., & Cheng, X. (2024). On protecting the data privacy of large language models (llms): A survey. arXiv preprint arXiv:2403.05156","DOI":"10.1109\/ICMC60390.2024.00008"},{"key":"6930_CR169","doi-asserted-by":"crossref","unstructured":"Yan, J., Yadav, V., Li, S., Chen, L., Tang, Z., Wang, H., Srinivasan, V., Ren, X., & Jin, H. (2024). Backdooring instruction-tuned large language models with virtual prompt injection. In: Proceedings of the 2024 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies (Volume 1: Long Papers), 6065\u20136086","DOI":"10.18653\/v1\/2024.naacl-long.337"},{"key":"6930_CR170","unstructured":"Yang, Y., Yao, H., Yang, B., He, Y., Li, Y., Zhang, T., Qin, Z., & Ren, K. (2024). Tapi: Towards target-specific and adversarial prompt injection against code llms. arXiv preprint arXiv:2407.09164"},{"key":"6930_CR171","doi-asserted-by":"crossref","unstructured":"Yao, Y., Xu, X., & Liu, Y. (2024). Large language model unlearning. In: The Thirty-eighth Annual Conference on Neural Information Processing Systems","DOI":"10.52202\/079017-3346"},{"key":"6930_CR172","doi-asserted-by":"publisher","unstructured":"Ye, R., Wang, W., Chai, J., Li, D., Li, Z., Xu, Y., Du, Y., Wang, Y., & Chen, S. (2024). Openfedllm: Training large language models on decentralized private data via federated learning. In: Proceedings of the 30th ACM SIGKDD Conference on Knowledge Discovery and Data Mining. KDD \u201924, pp. 6137\u20136147. Association for Computing Machinery, New York, NY, USA. https:\/\/doi.org\/10.1145\/3637528.3671582","DOI":"10.1145\/3637528.3671582"},{"key":"6930_CR173","doi-asserted-by":"publisher","unstructured":"Yermilov, O., Raheja, V., Chernodub, A. (2023). Privacy- and utility-preserving NLP with anonymized data: A case study of pseudonymization. In: Proceedings of the 3rd Workshop on Trustworthy Natural Language Processing (TrustNLP 2023), pp. 232\u2013241. Association for Computational Linguistics, Toronto, Canada. https:\/\/doi.org\/10.18653\/v1\/2023.trustnlp-1.20","DOI":"10.18653\/v1\/2023.trustnlp-1.20"},{"key":"6930_CR174","unstructured":"Yi, J., Xie, Y., Zhu, B., Hines, K., Kiciman, E., Sun, G., Xie, X., & Wu, F. (2023). Benchmarking and defending against indirect prompt injection attacks on large language models. arXiv preprint arXiv:2312.14197"},{"key":"6930_CR175","unstructured":"Yu, J., Lin, X., & Xing, X. (2023). Gptfuzzer: Red teaming large language models with auto-generated jailbreak prompts. arXiv preprint arXiv:2309.10253"},{"key":"6930_CR176","unstructured":"Yu, D., Naik, S., Backurs, A., Gopi, S., Inan, H.A., Kamath, G., Kulkarni, J., Lee, Y.T., Manoel, A., Wutschitz, L., et al. (2022). Differentially private fine-tuning of language models. In: International Conference on Learning Representations (ICLR)"},{"key":"6930_CR177","unstructured":"Yuan, W., Pang, R.Y., Cho, K., Li, X., Sukhbaatar, S., Xu, J., & Weston, J.E. (2024). Self-rewarding language models. In: Forty-first International Conference on Machine Learning"},{"key":"6930_CR178","doi-asserted-by":"crossref","unstructured":"Yuan, H., Yuan, Z., Tan, C., Wang, W., Huang, S., & Huang, F. (2024). Rrhf: Rank responses to align language models with human feedback. Advances in Neural Information Processing Systems 36","DOI":"10.52202\/075280-0482"},{"key":"6930_CR179","unstructured":"Zeng, W., Liu, Y., Mullins, R., Peran, L., Fernandez, J., Harkous, H., Narasimhan, K., Proud, D., Kumar, P., Radharapu, B., et al. (2024). Shieldgemma: Generative ai content moderation based on gemma. arXiv preprint arXiv:2407.21772"},{"key":"6930_CR180","doi-asserted-by":"publisher","unstructured":"Zhan, Q., Liang, Z., Ying, Z., & Kang, D. (2024). InjecAgent: Benchmarking indirect prompt injections in tool-integrated large language model agents. In: Findings of the Association for Computational Linguistics ACL 2024, pp. 10471\u201310506. Association for Computational Linguistics, Bangkok, Thailand and virtual meeting. https:\/\/doi.org\/10.18653\/v1\/2024.findings-acl.624","DOI":"10.18653\/v1\/2024.findings-acl.624"},{"key":"6930_CR181","doi-asserted-by":"crossref","unstructured":"Zhang, W., Gui, L., Procter, R., & He, Y. (2024). Multi-layer ranking with large language models for news source recommendation. In: Proceedings of the 47th International ACM SIGIR Conference on Research and Development in Information Retrieval, 2537\u20132542","DOI":"10.1145\/3626772.3657966"},{"key":"6930_CR182","unstructured":"Zhang, L., Li, B., Thekumparampil, K.K., Oh, S., & He, N. (2024). Dpzero: private fine-tuning of language models without backpropagation. In: Proceedings of the 41st International Conference on Machine Learning, 59210\u201359246"},{"key":"6930_CR183","unstructured":"Zhang, X., Zhang, C., Li, T., Huang, Y., Jia, X., Hu, M., Zhang, J., Liu, Y., Ma, S., & Shen, C. (2023). Jailguard: A universal detection framework for llm prompt-based attacks. arXiv preprint arXiv:2312.10766"},{"key":"6930_CR184","doi-asserted-by":"crossref","unstructured":"Zhang, Z., Zhang, X., Zhang, Y., Zhang, L.Y., Chen, C., Hu, S., Gill, A., & Pan, S. (2024). Large language model watermark stealing with mixed integer programming. arXiv preprint arXiv:2405.19677","DOI":"10.1109\/ACSAC63791.2024.00021"},{"key":"6930_CR185","doi-asserted-by":"publisher","unstructured":"Zhao, S., Gan, L., Luu, A.T., Fu, J., Lyu, L., Jia, M., & Wen, J. (2024). Defending against weight-poisoning backdoor attacks for parameter-efficient fine-tuning. In: Findings of the Association for Computational Linguistics: NAACL 2024, pp. 3421\u20133438. Association for Computational Linguistics, Mexico City, Mexico. https:\/\/doi.org\/10.18653\/v1\/2024.findings-naacl.217","DOI":"10.18653\/v1\/2024.findings-naacl.217"},{"key":"6930_CR186","unstructured":"Zhao, Y., Joshi, R., Liu, T., Khalman, M., Saleh, M., & Liu, P.J. (2023). Slic-hf: Sequence likelihood calibration with human feedback. arXiv preprint arXiv:2305.10425"},{"key":"6930_CR187","doi-asserted-by":"publisher","unstructured":"Zhao, S., Wen, J., Luu, A., Zhao, J., & Fu, J. (2023). Prompt as triggers for backdoor attack: Examining the vulnerability in language models. In: Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing, 12303\u201312317. Association for Computational Linguistics, Singapore. https:\/\/doi.org\/10.18653\/v1\/2023.emnlp-main.757","DOI":"10.18653\/v1\/2023.emnlp-main.757"},{"key":"6930_CR188","unstructured":"Zhao, W.X., Zhou, K., Li, J., Tang, T., Wang, X., Hou, Y., Min, Y., Zhang, B., Zhang, J., Dong, Z., et al. (2023). A survey of large language models. arXiv preprint arXiv:2303.18223"},{"key":"6930_CR189","unstructured":"Zheng, M., Pei, J., & Jurgens, D. (2023). Is\u201c a helpful assistant\u201d the best role for large language models? a systematic evaluation of social roles in system prompts. arXiv preprint arXiv:2311.10054"},{"key":"6930_CR190","unstructured":"Zheng, C., Yin, F., Zhou, H., Meng, F., Zhou, J., Chang, K.-W., Huang, M., & Peng, N. (2024). On prompt-driven safeguarding for large language models. In: Forty-first International Conference on Machine Learning"},{"key":"6930_CR191","doi-asserted-by":"crossref","unstructured":"Zheng, J., Zhang, H., Wang, L., Qiu, W., Zheng, H., & Zheng, Z. (2024). Safely learning with private data: A federated learning framework for large language model. arXiv preprint arXiv:2406.14898","DOI":"10.18653\/v1\/2024.emnlp-main.303"},{"key":"6930_CR192","doi-asserted-by":"crossref","unstructured":"Zhou, Y., He, B., & Sun, L. (2024). Humanizing machine-generated content: Evading ai-text detection through adversarial attack. In: Proceedings of the 2024 Joint International Conference on Computational Linguistics, Language Resources and Evaluation (LREC-COLING 2024), 8427\u20138437","DOI":"10.63317\/5ocyse7sux8y"},{"key":"6930_CR193","doi-asserted-by":"crossref","unstructured":"Zhu, K., Wang, J., Zhou, J., Wang, Z., Chen, H., Wang, Y., Yang, L., Ye, W., Gong, N.Z., Zhang, Y., et al. (2023). Promptbench: Towards evaluating the robustness of large language models on adversarial prompts. arXiv preprint arXiv:2306.04528","DOI":"10.1145\/3689217.3690621"},{"key":"6930_CR194","unstructured":"Zhu, S., Zhang, R., An, B., Wu, G., Barrow, J., Wang, Z., Huang, F., Nenkova, A., & Sun, T. (2024). AutoDAN: Interpretable gradient-based adversarial attacks on large language models. In: First Conference on Language Modeling. https:\/\/openreview.net\/forum?id=INivcBeIDK"},{"key":"6930_CR195","unstructured":"Zou, A., Wang, Z., Kolter, J.Z., & Fredrikson, M. (2023). Universal and transferable adversarial attacks on aligned language models. arXiv preprint arXiv:2307.15043"}],"container-title":["Machine Learning"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10994-025-06930-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10994-025-06930-x","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10994-025-06930-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,11]],"date-time":"2026-07-11T07:13:36Z","timestamp":1783754016000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10994-025-06930-x"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,5,23]]},"references-count":195,"journal-issue":{"issue":"6","published-print":{"date-parts":[[2026,6]]}},"alternative-id":["6930"],"URL":"https:\/\/doi.org\/10.1007\/s10994-025-06930-x","relation":{},"ISSN":["0885-6125","1573-0565"],"issn-type":[{"value":"0885-6125","type":"print"},{"value":"1573-0565","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,5,23]]},"assertion":[{"value":"1 March 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"28 July 2025","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"31 October 2025","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"23 May 2026","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no Conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}],"article-number":"133"}}