{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,26]],"date-time":"2026-02-26T08:16:52Z","timestamp":1772093812794,"version":"3.50.1"},"publisher-location":"Singapore","reference-count":27,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819549719","type":"print"},{"value":"9789819549726","type":"electronic"}],"license":[{"start":{"date-parts":[[2025,11,27]],"date-time":"2025-11-27T00:00:00Z","timestamp":1764201600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,11,27]],"date-time":"2025-11-27T00:00:00Z","timestamp":1764201600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026]]},"DOI":"10.1007\/978-981-95-4972-6_32","type":"book-chapter","created":{"date-parts":[[2025,11,26]],"date-time":"2025-11-26T08:08:18Z","timestamp":1764144498000},"page":"414-425","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Guardrail Guided Policy Optimisation: Learning Disentangled Safety Constraints"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0009-0008-4900-2660","authenticated-orcid":false,"given":"Jaymari","family":"Chua","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3119-4763","authenticated-orcid":false,"given":"Chen","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5839-3765","authenticated-orcid":false,"given":"Liming","family":"Zhu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4149-839X","authenticated-orcid":false,"given":"Lina","family":"Yao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,11,27]]},"reference":[{"key":"32_CR1","doi-asserted-by":"crossref","unstructured":"Abbeel, P., Ng, A.Y.: Apprenticeship learning via inverse reinforcement learning. In: Proceedings of the Twenty-First International Conference on Machine Learning, p.\u00a01 (2004)","DOI":"10.1145\/1015330.1015430"},{"key":"32_CR2","unstructured":"Achiam, J., Held, D., Tamar, A., Abbeel, P.: Constrained policy optimization. In: International Conference on Machine Learning, pp. 22\u201331. PMLR (2017)"},{"key":"32_CR3","unstructured":"Altman, E.: Constrained Markov decision processes. In: Stochastic Modeling Series, vol.\u00a07, pp. 1\u2013242. CRC Press (1999)"},{"key":"32_CR4","doi-asserted-by":"crossref","unstructured":"Bender, E.M., Koller, A.: Climbing towards NLU: on meaning, form, and understanding in the age of data. In: Proceedings of the 58th Annual Meeting of the Association for Computational Linguistics, pp. 5185\u20135198 (2020)","DOI":"10.18653\/v1\/2020.acl-main.463"},{"key":"32_CR5","unstructured":"Casper, S., et al.: Open problems and fundamental limitations of reinforcement learning from human feedback. arXiv preprint arXiv:2307.15217 (2023)"},{"key":"32_CR6","unstructured":"Dai, J., et al.: Safe RLHF: safe reinforcement learning from human feedback. arXiv preprint arXiv:2310.12773 (2023)"},{"key":"32_CR7","unstructured":"Dziri, N., et al.: Faith and fate: limits of transformers on compositionality. arXiv preprint arXiv:2305.18654 (2023)"},{"key":"32_CR8","unstructured":"Gao, L., Schulman, J., Hilton, J.: Scaling laws for reward model overoptimization. arXiv preprint arXiv:2210.10760 (2022)"},{"key":"32_CR9","unstructured":"Ho, J., Ermon, S.: Generative adversarial imitation learning. In: Advances in Neural Information Processing Systems, vol.\u00a029 (2016)"},{"key":"32_CR10","unstructured":"Kirk, M., Rttger, P., Gowal, S., Bunel, R., Gal, Y.: Understanding reward model overoptimization from causal and mechanistic perspectives. arXiv preprint arXiv:2310.19960 (2023)"},{"key":"32_CR11","unstructured":"Lampinen, A.K., et al.: Can language models learn from explanations in context? In: Proceedings of the 60th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), pp. 6858\u20136884 (2022)"},{"key":"32_CR12","unstructured":"Li, J., et al.: Inferring rewards from language explanations. In: Proceedings of the 2023 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies: Findings, pp. 2269\u20132283 (2023)"},{"key":"32_CR13","unstructured":"Liu, S., Zhu, M.: Distributed inverse constrained reinforcement learning for multi-agent systems. In: Koyejo, S., Mohamed, S., Agarwal, A., Belgrave, D., Cho, K., Oh, A. (eds.) Advances in Neural Information Processing Systems, vol.\u00a035, pp. 34050\u201334063 (2022)"},{"key":"32_CR14","unstructured":"Ma, C., et al.: Following instructions with preferences: aligning language models via constrained preference optimization. arXiv preprint arXiv:2310.14915 (2023)"},{"key":"32_CR15","unstructured":"Malik, S., Anwar, U., Aghasi, A., Ahmed, A.: Inverse constrained reinforcement learning. In: Meila, M., Zhang, T. (eds.) Proceedings of the 38th International Conference on Machine Learning. Proceedings of Machine Learning Research, vol.\u00a0139, pp. 7390\u20137399. PMLR (2021). https:\/\/proceedings.mlr.press\/v139\/malik21a.html"},{"key":"32_CR16","unstructured":"Ng, A.Y., Russell, S., et\u00a0al.: Algorithms for inverse reinforcement learning. In: ICML, vol.\u00a01, p.\u00a02 (2000)"},{"key":"32_CR17","unstructured":"Perez, E., et al.: Red teaming language models to reduce harms: methods, scaling behaviors, and lessons learned. In: Proceedings of the 60th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), pp. 3150\u20133179 (2022)"},{"key":"32_CR18","unstructured":"Ramamurthy, A., Patel, H., Iyer, S., Chen, V., Baral, C.: Is the majority really harder? A parsimonious debugging framework for in-context learning. In: Thirty-Seventh Conference on Neural Information Processing Systems (2023)"},{"key":"32_CR19","unstructured":"Ray, A., Achiam, J., Amodei, D.: Benchmarking safe exploration in deep reinforcement learning. In: Thirty-Third Conference on Neural Information Processing Systems (2019)"},{"key":"32_CR20","unstructured":"R\u00f6ttger, P., Bunel, R., Gal, Y., Modgil, S.: Inference-time policy adapters: resisting mode collapse by adapting to evolving rewards. In: Thirty-Eighth AAAI Conference on Artificial Intelligence (2024)"},{"key":"32_CR21","unstructured":"Scobee, D.R.R., Sastry, S.S.: Maximum likelihood constraint inference for inverse reinforcement learning. In: International Conference on Learning Representations (2020). https:\/\/openreview.net\/forum?id=BJliakStvH"},{"key":"32_CR22","unstructured":"Sun, H., van\u00a0der Schaar, M.: Inverse-rlignment: inverse reinforcement learning from demonstrations for LLM alignment. arXiv preprint arXiv:2405.15624 (2024)"},{"key":"32_CR23","unstructured":"Wei, A., et al.: Jailbroken: how does LLM safety training fail? arXiv preprint arXiv:2311.17614 (2023)"},{"key":"32_CR24","unstructured":"Xu, R., Lu, S., Zhou, Y., Li, Z., Chai, J.: Preference-aware task adaptation for reinforcement learning. arXiv preprint arXiv:2304.02480 (2023)"},{"key":"32_CR25","doi-asserted-by":"publisher","unstructured":"Yang, B., Li, Z., Li, S.: Worry no more: a safety-enhanced model-based approach for safe reinforcement learning in autonomous driving. In: Proceedings of the 30th ACM International Conference on Information & Knowledge Management, pp. 2432\u20132441. Association for Computing Machinery, New York (2021). https:\/\/doi.org\/10.1145\/3474085.3475212","DOI":"10.1145\/3474085.3475212"},{"key":"32_CR26","unstructured":"Yue, B., Li, J., Liu, G.: Provably efficient exploration in inverse constrained reinforcement learning. In: Proceedings of the 42nd International Conference on Machine Learning (2025). https:\/\/icml.cc\/virtual\/2025\/poster\/44588"},{"key":"32_CR27","unstructured":"Ziebart, B.D., Maas, A.L., Bagnell, J.A., Dey, A.K.: Maximum entropy inverse reinforcement learning. In: Proceedings of the 23rd National Conference on Artificial Intelligence, vol.\u00a03, pp. 1433\u20131438 (2008)"}],"container-title":["Lecture Notes in Computer Science","AI 2025: Advances in Artificial Intelligence"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-95-4972-6_32","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,2,26]],"date-time":"2026-02-26T07:23:37Z","timestamp":1772090617000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-95-4972-6_32"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,11,27]]},"ISBN":["9789819549719","9789819549726"],"references-count":27,"URL":"https:\/\/doi.org\/10.1007\/978-981-95-4972-6_32","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,11,27]]},"assertion":[{"value":"27 November 2025","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"AI","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Australasian Joint Conference on Artificial Intelligence","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Canberra, ACT","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Australia","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2025","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"1 December 2025","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"5 December 2025","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"38","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"ausai2025","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/ajcai2025.org\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}