{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,18]],"date-time":"2026-07-18T11:02:34Z","timestamp":1784372554660,"version":"3.55.0"},"publisher-location":"Singapore","reference-count":19,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819235193","type":"print"},{"value":"9789819235209","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,7,19]],"date-time":"2026-07-19T00:00:00Z","timestamp":1784419200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0"},{"start":{"date-parts":[[2026,7,19]],"date-time":"2026-07-19T00:00:00Z","timestamp":1784419200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2027]]},"DOI":"10.1007\/978-981-92-3520-9_13","type":"book-chapter","created":{"date-parts":[[2026,7,18]],"date-time":"2026-07-18T10:03:46Z","timestamp":1784369026000},"page":"157-169","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["S3L: Safety Subspace Structured LoRA with Risk-Aware Expert Routing for LLM Safety Alignment"],"prefix":"10.1007","author":[{"given":"Xinghao","family":"Wang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yuke","family":"Ma","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xi","family":"Chen","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Wei","family":"Guo","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,7,19]]},"reference":[{"key":"13_CR1","unstructured":"Zou, A., Wang, Z., Carlini, N., Nasr, M., Kolter, J.Z., Fredrikson, M.: Universal and transferable adversarial attacks on aligned language models. arXiv2307.15043 (2023)"},{"key":"13_CR2","unstructured":"Liu, X., Xu, N., Chen, M., Xiao, C.: AutoDAN: generating stealthy jailbreak prompts on aligned large language models. In: ICLR (2024)"},{"key":"13_CR3","unstructured":"Hu, E.J., et al.: LoRA: low-rank adaptation of large language models. In: ICLR (2022)"},{"key":"13_CR4","unstructured":"Qi, X., et al.: Fine-tuning aligned language models compromises safety, even when users do not intend to! In: ICLR (2024)"},{"key":"13_CR5","unstructured":"Robey, A., Wong, E., Hassani, H., Pappas, G.J.: SmoothLLM: defending large language models against jailbreaking attacks. arXiv2310.03684 (2023)"},{"key":"13_CR6","first-page":"1486","volume":"5","author":"Y Xie","year":"2023","unstructured":"Xie, Y., et al.: Defending ChatGPT against jailbreak attack via self-reminders. NatMI 5, 1486\u20131496 (2023)","journal-title":"NatMI"},{"key":"13_CR7","unstructured":"Mazeika, M., et al.: HarmBench: a standardized evaluation framework for automated red teaming and robust refusal. In: Proceedings of ICML. PMLR, vol. 235, pp. 35181\u201335224. PMLR (2024)"},{"key":"13_CR8","doi-asserted-by":"crossref","unstructured":"Chao, P., et al.: JailbreakBench: an open robustness benchmark for jailbreaking large language models. In: NeurIPS, vol. 37 (2024)","DOI":"10.52202\/079017-1745"},{"key":"13_CR9","doi-asserted-by":"crossref","unstructured":"Souly, A., et al.: A StrongREJECT for empty jailbreaks. In: NeurIPS, vol. 37 (2024)","DOI":"10.52202\/079017-3984"},{"key":"13_CR10","unstructured":"Wu, X., Huang, S., Wei, F.: Mixture of LoRA experts. In: ICLR (2024)"},{"key":"13_CR11","doi-asserted-by":"crossref","unstructured":"Dou, S., et al.: LoRAMoE: alleviating world knowledge forgetting in large language models via MoE-style plugin. In: Proceedings of ACL, pp.1932\u20131945. ACL (2024)","DOI":"10.18653\/v1\/2024.acl-long.106"},{"key":"13_CR12","doi-asserted-by":"crossref","unstructured":"R\u00f6ttger, P., Kirk, H.R., Vidgen, B., Attanasio, G., Bianchi, F., Hovy, D.: XSTest: a test suite for identifying exaggerated safety behaviours in large language models. In: Proceedings of the NAACL-HLT, pp.5377\u20135400 (2024)","DOI":"10.18653\/v1\/2024.naacl-long.301"},{"key":"13_CR13","unstructured":"Cui, J., Chiang, W.L., Stoica, I., Hsieh, C.J.: OR-bench: an over-refusal benchmark for large language models. In: Proceedings of the ICML. PMLR, vol. 267, pp. 11515\u201311542. PMLR (2025)"},{"key":"13_CR14","unstructured":"Zhang, Z., Xu, W., Wu, F., Reddy, C.K.: FalseReject: a resource for improving contextual safety and mitigating over-refusals in LLMs via structured reasoning. In: The Second Conference on Language Modeling (2025)"},{"key":"13_CR15","unstructured":"Turner, A.M., et al.: Steering language models with activation engineering. arXiv2308.10248 (2023)"},{"key":"13_CR16","unstructured":"Tigges, C., Hollinsworth, O.J., Geiger, A., Nanda, N.: Linear representations of sentiment in large language models. arXiv2310.15154 (2023)"},{"key":"13_CR17","unstructured":"Zou, A., et al.: Representation engineering: a top-down approach to AI transparency. arXiv2310.01405 (2023)"},{"issue":"1","key":"13_CR18","doi-asserted-by":"publisher","first-page":"2134","DOI":"10.1038\/s41467-018-04608-8","volume":"9","author":"A Abid","year":"2018","unstructured":"Abid, A., Zhang, M.J., Bagaria, V.K., Zou, J.: Exploring patterns enriched in a dataset with contrastive principal component analysis. Nat. Commun. 9(1), 2134 (2018)","journal-title":"Nat. Commun."},{"key":"13_CR19","unstructured":"Tunstall, L., et al.: Zephyr: Direct distillation of LM alignment. In: Proceedings of the COLM (2024)"}],"container-title":["Lecture Notes in Computer Science","Advanced Intelligent Computing Technology and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-92-3520-9_13","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,18]],"date-time":"2026-07-18T10:03:48Z","timestamp":1784369028000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-92-3520-9_13"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,19]]},"ISBN":["9789819235193","9789819235209"],"references-count":19,"URL":"https:\/\/doi.org\/10.1007\/978-981-92-3520-9_13","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,7,19]]},"assertion":[{"value":"19 July 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ICIC","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Intelligent Computing","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Toronto","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Canada","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2026","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"22 July 2026","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"26 July 2026","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"22","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"icic2026a","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"http:\/\/www.ic-icc.cn\/2026\/index.htm","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}