{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,6,21]],"date-time":"2025-06-21T17:40:11Z","timestamp":1750527611117,"version":"3.41.0"},"publisher-location":"Singapore","reference-count":20,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819681969","type":"print"},{"value":"9789819681976","type":"electronic"}],"license":[{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-981-96-8197-6_26","type":"book-chapter","created":{"date-parts":[[2025,6,21]],"date-time":"2025-06-21T17:19:15Z","timestamp":1750526355000},"page":"349-361","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Enhancing AI Safety Through the\u00a0Fusion of\u00a0Low Rank Adapters"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-6714-9382","authenticated-orcid":false,"given":"Satya Swaroop","family":"Gudipudi","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-4229-8588","authenticated-orcid":false,"given":"Sreeram","family":"Vipparla","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-6642-4568","authenticated-orcid":false,"given":"Harpreet","family":"Singh","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-5193-8135","authenticated-orcid":false,"given":"Shashwat","family":"Goel","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5082-2078","authenticated-orcid":false,"given":"Ponnurangam","family":"Kumaraguru","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,6,22]]},"reference":[{"key":"26_CR1","unstructured":"OpenAI, Achiam, J., Adler, S., Agarwal, S., et al.: GPT-4 Technical Report (2024). https:\/\/arxiv.org\/abs\/2303.08774"},{"key":"26_CR2","unstructured":"Brown, T. et al.: Language models are few-shot learners. In: NeurIPS , pp. 1877\u20131901 (2020). https:\/\/proceedings.neurips.cc\/paper\/2020\/file\/1457c0d6bfcb4967418bfb8ac142f64a-Paper.pdf"},{"key":"26_CR3","unstructured":"Mangrulkar, S. et al.: PEFT: state-of-the-art parameter-efficient fine-tuning methods. https:\/\/github.com\/huggingface\/peft. Accessed 25 Oct 2023"},{"key":"26_CR4","unstructured":"Hu, E.J. et al.: LoRA: low-rank adaptation of large language models. In: International Conference on Learning Representations (ICLR) (2022). https:\/\/openreview.net\/forum?id=nZeVKeeFYf9"},{"key":"26_CR5","unstructured":"Christiano, P.F. et al.: Deep reinforcement learning from human preferences. In: NeurIPS (2017). https:\/\/proceedings.neurips.cc\/paper_files\/paper\/2017\/file\/d5e2c0adad503c91f91df240d0cd4e49-Paper.pdf"},{"key":"26_CR6","unstructured":"Jain, S. et al.: Mechanistically analyzing the effects of fine-tuning on procedurally defined tasks. In: ICLR (2024). https:\/\/openreview.net\/forum?id=A0HKeKl4Nl"},{"key":"26_CR7","unstructured":"Qi, X. et al.: Fine-tuning aligned language models compromises safety, even when users do not intend to! In: The Twelfth International Conference on Learning Representations (ICLR) (2024). https:\/\/openreview.net\/forum?id=hTEGyKf0dZ"},{"key":"26_CR8","doi-asserted-by":"publisher","unstructured":"Zhan, Q., et al.: Removing RLHF protections in GPT-4 via fine-tuning. In: Proceedings of the NAACL-HLT, vol. 2, pp. 681\u2013687, Mexico City (2024). https:\/\/doi.org\/10.18653\/v1\/2024.naacl-short.59, https:\/\/aclanthology.org\/2024.naacl-short.59","DOI":"10.18653\/v1\/2024.naacl-short.59"},{"key":"26_CR9","unstructured":"Bianchi, F. et al.: Safety-Tuned LLaMAs: lessons from improving the safety of large language models that follow instructions. In: ICLR (2024). https:\/\/openreview.net\/forum?id=gT5hALch9z"},{"key":"26_CR10","unstructured":"Wallace, E. et al.: The instruction hierarchy: training LLMs to prioritize privileged instructions. ArXiv (2024). cs.CR, eprint 2404.13208. https:\/\/arxiv.org\/abs\/2404.13208"},{"key":"26_CR11","unstructured":"R\u00f6ttger, P. et al.: XSTest: identifying exaggerated safety behaviours in LLMs. NAACL-HLT, pp. 5377\u20135400 (2024). https:\/\/aclanthology.org\/2024.naacl-long.301"},{"key":"26_CR12","unstructured":"Zou, A. et al.: Universal and transferable adversarial attacks on aligned language models (2023). https:\/\/arxiv.org\/abs\/2307.15043"},{"key":"26_CR13","unstructured":"Eiras, F., et al.: mimicking user data: mitigating fine-tuning risks in closed LLMs (2024). https:\/\/arxiv.org\/abs\/2406.10288"},{"key":"26_CR14","unstructured":"Hendrycks, D. et al.: Measuring massive multitask language understanding. In: International Conference on Learning Representations (ICLR) (2021). https:\/\/openreview.net\/forum?id=d7KBjmI3GmQ"},{"key":"26_CR15","unstructured":"Touvron, H. et al.: Llama 2: open foundation and fine-tuned chat models. arXiv preprint arXiv:2307.09288 (2023). https:\/\/arxiv.org\/abs\/2307.09288"},{"key":"26_CR16","unstructured":"QWEN Team: Introducing Qwen1.5 (2024). https:\/\/qwenlm.github.io\/blog\/qwen1.5\/"},{"key":"26_CR17","unstructured":"Grattafiori, A. et al.: The Llama 3 herd of models. arXiv preprint arXiv:2407.21783 (2024). https:\/\/arxiv.org\/abs\/2407.21783"},{"key":"26_CR18","unstructured":"Zhang, J. et al.: Composing parameter-efficient modules with arithmetic operations. In: Proc. of NeurIPS Art. No. 552, pp. 1\u201322 (2024)"},{"key":"26_CR19","unstructured":"Wu, X. et al.: Mixture of LoRA experts. In: The Twelfth International Conference on Learning Representations, (2024). https:\/\/openreview.net\/forum?id=uWvKBCYh4S"},{"key":"26_CR20","unstructured":"Hsu, C.-Y. et al.: Safe LoRA: reducing safety risks when fine-tuning large language models. arXiv:2405.16833 (2024). https:\/\/arxiv.org\/abs\/2405.16833"}],"container-title":["Lecture Notes in Computer Science","Trends and Applications in Knowledge Discovery and Data Mining"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-96-8197-6_26","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,21]],"date-time":"2025-06-21T17:19:20Z","timestamp":1750526360000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-96-8197-6_26"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025]]},"ISBN":["9789819681969","9789819681976"],"references-count":20,"URL":"https:\/\/doi.org\/10.1007\/978-981-96-8197-6_26","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025]]},"assertion":[{"value":"22 June 2025","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"We declare no conflict of interest currently.","order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Disclosure of Interests"}},{"value":"PAKDD","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Pacific-Asia Conference on Knowledge Discovery and Data Mining","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Sydney, NSW","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Australia","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2025","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"10 June 2025","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"13 June 2025","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"pakdd2025","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/pakdd2025.org\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}