{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,2]],"date-time":"2026-07-02T16:30:50Z","timestamp":1783009850559,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":77,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,12,2]],"date-time":"2024-12-02T00:00:00Z","timestamp":1733097600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,12,2]]},"DOI":"10.1145\/3658644.3690217","type":"proceedings-article","created":{"date-parts":[[2024,12,9]],"date-time":"2024-12-09T12:19:20Z","timestamp":1733746760000},"page":"1136-1150","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":8,"title":["A Causal Explainable Guardrails for Large Language Models"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-6075-1816","authenticated-orcid":false,"given":"Zhixuan","family":"Chu","sequence":"first","affiliation":[{"name":"Zhejiang University &amp; State Key Laboratory of Blockchain and Data Security, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-2938-357X","authenticated-orcid":false,"given":"Yan","family":"Wang","sequence":"additional","affiliation":[{"name":"Ant Group, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9263-7011","authenticated-orcid":false,"given":"Longfei","family":"Li","sequence":"additional","affiliation":[{"name":"Ant Group, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5804-3279","authenticated-orcid":false,"given":"Zhibo","family":"Wang","sequence":"additional","affiliation":[{"name":"Zhejiang University, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7872-6969","authenticated-orcid":false,"given":"Zhan","family":"Qin","sequence":"additional","affiliation":[{"name":"Zhejiang University, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3441-6277","authenticated-orcid":false,"given":"Kui","family":"Ren","sequence":"additional","affiliation":[{"name":"Zhejiang University, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2024,12,9]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Understanding intermediate layers using linear classifier probes. arXiv preprint arXiv:1610.01644","author":"Alain Guillaume","year":"2016","unstructured":"Guillaume Alain and Yoshua Bengio. 2016. Understanding intermediate layers using linear classifier probes. arXiv preprint arXiv:1610.01644 (2016)."},{"key":"e_1_3_2_1_2_1","volume-title":"The internal state of an llm knows when its lying. arXiv preprint arXiv:2304.13734","author":"Azaria Amos","year":"2023","unstructured":"Amos Azaria and Tom Mitchell. 2023. The internal state of an llm knows when its lying. arXiv preprint arXiv:2304.13734 (2023)."},{"key":"e_1_3_2_1_3_1","volume-title":"Defending pre-trained language models from adversarial word substitutions without performance sacrifice. arXiv preprint arXiv:2105.14553","author":"Bao Rongzhou","year":"2021","unstructured":"Rongzhou Bao, Jiayi Wang, and Hai Zhao. 2021. Defending pre-trained language models from adversarial word substitutions without performance sacrifice. arXiv preprint arXiv:2105.14553 (2021)."},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1162\/coli_a_00422"},{"key":"e_1_3_2_1_5_1","volume-title":"Eliciting latent predictions from transformers with the tuned lens. arXiv preprint arXiv:2303.08112","author":"Belrose Nora","year":"2023","unstructured":"Nora Belrose, Zach Furman, Logan Smith, Danny Halawi, Igor Ostrovsky, Lev McKinney, Stella Biderman, and Jacob Steinhardt. 2023. Eliciting latent predictions from transformers with the tuned lens. arXiv preprint arXiv:2303.08112 (2023)."},{"key":"e_1_3_2_1_6_1","unstructured":"Tolga Bolukbasi Kai-Wei Chang James Y Zou Venkatesh Saligrama and Adam T Kalai. 2016. Man is to computer programmer as woman is to homemaker? debiasing word embeddings. In Advances in neural information processing systems. 4349--4357."},{"key":"e_1_3_2_1_7_1","unstructured":"Tom Brown Benjamin Mann Nick Ryder Melanie Subbiah Jared D Kaplan Prafulla Dhariwal Arvind Neelakantan Pranav Shyam Girish Sastry Amanda Askell et al. 2020. Language models are few-shot learners. Advances in neural information processing systems Vol. 33 (2020) 1877--1901."},{"key":"e_1_3_2_1_8_1","volume-title":"Discovering latent knowledge in language models without supervision. arXiv preprint arXiv:2212.03827","author":"Burns Collin","year":"2022","unstructured":"Collin Burns, Haotian Ye, Dan Klein, and Jacob Steinhardt. 2022. Discovering latent knowledge in language models without supervision. arXiv preprint arXiv:2212.03827 (2022)."},{"key":"e_1_3_2_1_9_1","volume-title":"Can prompt probe pretrained language models? understanding the invisible risks from a causal view. arXiv preprint arXiv:2203.12258","author":"Cao Boxi","year":"2022","unstructured":"Boxi Cao, Hongyu Lin, Xianpei Han, Fangchao Liu, and Le Sun. 2022. Can prompt probe pretrained language models? understanding the invisible risks from a causal view. arXiv preprint arXiv:2203.12258 (2022)."},{"key":"e_1_3_2_1_10_1","volume-title":"Jailbreaking Black Box Large Language Models in Twenty Queries. CoRR","author":"Chao Patrick","year":"2023","unstructured":"Patrick Chao, Alexander Robey, Edgar Dobriban, Hamed Hassani, George J. Pappas, and Eric Wong. 2023. Jailbreaking Black Box Large Language Models in Twenty Queries. CoRR, Vol. abs\/2310.08419 (2023)."},{"key":"e_1_3_2_1_11_1","volume-title":"INSIDE: LLMs' Internal States Retain the Power of Hallucination Detection. arXiv preprint arXiv:2402.03744","author":"Chen Chao","year":"2024","unstructured":"Chao Chen, Kai Liu, Ze Chen, Yi Gu, Yue Wu, Mingyuan Tao, Zhihang Fu, and Jieping Ye. 2024. INSIDE: LLMs' Internal States Retain the Power of Hallucination Detection. arXiv preprint arXiv:2402.03744 (2024)."},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/TKDE.2024.3352575"},{"key":"e_1_3_2_1_13_1","volume-title":"Vicuna: An open-source chatbot impressing gpt-4 with 90%* chatgpt quality. See https:\/\/vicuna. lmsys. org (accessed","author":"Chiang Wei-Lin","year":"2023","unstructured":"Wei-Lin Chiang, Zhuohan Li, Zi Lin, Ying Sheng, Zhanghao Wu, Hao Zhang, Lianmin Zheng, Siyuan Zhuang, Yonghao Zhuang, Joseph E Gonzalez, et al. 2023. Vicuna: An open-source chatbot impressing gpt-4 with 90%* chatgpt quality. See https:\/\/vicuna. lmsys. org (accessed 14 April 2023) (2023)."},{"key":"e_1_3_2_1_14_1","volume-title":"Causal Interventional Prediction System for Robust and Explainable Effect Forecasting. arXiv preprint arXiv:2407.19688","author":"Chu Zhixuan","year":"2024","unstructured":"Zhixuan Chu, Hui Ding, Guang Zeng, Shiyu Wang, and Yiming Li. 2024. Causal Interventional Prediction System for Robust and Explainable Effect Forecasting. arXiv preprint arXiv:2407.19688 (2024)."},{"key":"e_1_3_2_1_15_1","unstructured":"Zhixuan Chu Huaiyu Guo Xinyuan Zhou Yijia Wang Fei Yu Hong Chen Wanqing Xu Xin Lu Qing Cui Longfei Li et al. 2023. Data-centric financial large language models. arXiv preprint arXiv:2310.17784 (2023)."},{"key":"e_1_3_2_1_16_1","volume-title":"Causal Effect Estimation: Recent Progress, Challenges, and Opportunities. Machine Learning for Causal Inference","author":"Chu Zhixuan","year":"2023","unstructured":"Zhixuan Chu and Sheng Li. 2023. Causal Effect Estimation: Recent Progress, Challenges, and Opportunities. Machine Learning for Causal Inference (2023), 79--100."},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1145\/3447548.3467302"},{"key":"e_1_3_2_1_18_1","volume-title":"Conference on health, inference, and learning. PMLR, 79--91","author":"Chu Zhixuan","year":"2022","unstructured":"Zhixuan Chu, Stephen L Rathbun, and Sheng Li. 2022. Multi-task adversarial learning for treatment effect estimation in basket trials. In Conference on health, inference, and learning. PMLR, 79--91."},{"key":"e_1_3_2_1_19_1","volume-title":"Llm-guided multi-view hypergraph learning for human-centric explainable recommendation. arXiv preprint arXiv:2401.08217","author":"Chu Zhixuan","year":"2024","unstructured":"Zhixuan Chu, Yan Wang, Qing Cui, Longfei Li, Wenqing Chen, Sheng Li, Zhan Qin, and Kui Ren. 2024. Llm-guided multi-view hypergraph learning for human-centric explainable recommendation. arXiv preprint arXiv:2401.08217 (2024)."},{"key":"e_1_3_2_1_20_1","volume-title":"Professional Agents--Evolving Large Language Models into Autonomous Experts with Human-Level Competencies. arXiv preprint arXiv:2402.03628","author":"Chu Zhixuan","year":"2024","unstructured":"Zhixuan Chu, Yan Wang, Feng Zhu, Lu Yu, Longfei Li, and Jinjie Gu. 2024. Professional Agents--Evolving Large Language Models into Autonomous Experts with Human-Level Competencies. arXiv preprint arXiv:2402.03628 (2024)."},{"key":"e_1_3_2_1_21_1","volume-title":"2024 d. Sora Detector: A Unified Hallucination Detection for Large Text-to-Video Models. arXiv preprint arXiv:2405.04180","author":"Chu Zhixuan","year":"2024","unstructured":"Zhixuan Chu, Lei Zhang, Yichen Sun, Siqiao Xue, Zhibo Wang, Zhan Qin, and Kui Ren. 2024 d. Sora Detector: A Unified Hallucination Detection for Large Text-to-Video Models. arXiv preprint arXiv:2405.04180 (2024)."},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1145\/3394486.3406460"},{"key":"e_1_3_2_1_23_1","volume-title":"Plug and play language models: A simple approach to controlled text generation. arXiv preprint arXiv:1912.02164","author":"Dathathri Sumanth","year":"2019","unstructured":"Sumanth Dathathri, Andrea Madotto, Janice Lan, Jane Hung, Eric Frank, Piero Molino, Jason Yosinski, and Rosanne Liu. 2019. Plug and play language models: A simple approach to controlled text generation. arXiv preprint arXiv:1912.02164 (2019)."},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1145\/3442188.3445924"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v36i11.21443"},{"key":"e_1_3_2_1_26_1","volume-title":"International conference on machine learning. PMLR, 1180--1189","author":"Ganin Yaroslav","year":"2015","unstructured":"Yaroslav Ganin and Victor Lempitsky. 2015. Unsupervised domain adaptation by backpropagation. In International conference on machine learning. PMLR, 1180--1189."},{"key":"e_1_3_2_1_27_1","volume-title":"Is chatgpt a good causal reasoner? a comprehensive evaluation. arXiv preprint arXiv:2305.07375","author":"Gao Jinglong","year":"2023","unstructured":"Jinglong Gao, Xiao Ding, Bing Qin, and Ting Liu. 2023. Is chatgpt a good causal reasoner? a comprehensive evaluation. arXiv preprint arXiv:2305.07375 (2023)."},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1145\/3637528.3671646"},{"key":"e_1_3_2_1_29_1","volume-title":"Language models represent space and time. arXiv preprint arXiv:2310.02207","author":"Gurnee Wes","year":"2023","unstructured":"Wes Gurnee and Max Tegmark. 2023. Language models represent space and time. arXiv preprint arXiv:2310.02207 (2023)."},{"key":"e_1_3_2_1_30_1","volume-title":"Toxigen: A large-scale machine-generated dataset for adversarial and implicit hate speech detection. arXiv preprint arXiv:2203.09509","author":"Hartvigsen Thomas","year":"2022","unstructured":"Thomas Hartvigsen, Saadia Gabriel, Hamid Palangi, Maarten Sap, Dipankar Ray, and Ece Kamar. 2022. Toxigen: A large-scale machine-generated dataset for adversarial and implicit hate speech detection. arXiv preprint arXiv:2203.09509 (2022)."},{"key":"e_1_3_2_1_31_1","volume-title":"Inspecting and editing knowledge representations in language models. arXiv preprint arXiv:2304.00740","author":"Hernandez Evan","year":"2023","unstructured":"Evan Hernandez, Belinda Z Li, and Jacob Andreas. 2023. Inspecting and editing knowledge representations in language models. arXiv preprint arXiv:2304.00740 (2023)."},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1145\/3623377"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.trustnlp-1.11"},{"key":"e_1_3_2_1_34_1","unstructured":"Lei Huang Weijiang Yu Weitao Ma Weihong Zhong Zhangyin Feng Haotian Wang Qianglong Chen Weihua Peng Xiaocheng Feng Bing Qin et al. 2023. A survey on hallucination in large language models: Principles taxonomy challenges and open questions. arXiv preprint arXiv:2311.05232 (2023)."},{"key":"e_1_3_2_1_35_1","unstructured":"C.J. Hutto. 2022. VADER-Sentiment-Analysis. https:\/\/github.com\/cjhutto\/vaderSentiment"},{"key":"e_1_3_2_1_36_1","volume-title":"Improving activation steering in language models with mean-centring. arXiv preprint arXiv:2312.03813","author":"Jorgensen Ole","year":"2023","unstructured":"Ole Jorgensen, Dylan Cope, Nandi Schoots, and Murray Shanahan. 2023. Improving activation steering in language models with mean-centring. arXiv preprint arXiv:2312.03813 (2023)."},{"key":"e_1_3_2_1_37_1","volume-title":"Wooyoung Kang, Byungseok Roh, and Hyunwoo J Kim.","author":"Ko Dohwan","year":"2023","unstructured":"Dohwan Ko, Ji Soo Lee, Wooyoung Kang, Byungseok Roh, and Hyunwoo J Kim. 2023. Large language models are temporal and causal reasoners for video question answering. arXiv preprint arXiv:2310.15747 (2023)."},{"key":"e_1_3_2_1_38_1","volume-title":"Rlaif: Scaling reinforcement learning from human feedback with ai feedback. arXiv preprint arXiv:2309.00267","author":"Lee Harrison","year":"2023","unstructured":"Harrison Lee, Samrat Phatale, Hassan Mansoor, Kellie Lu, Thomas Mesnard, Colton Bishop, Victor Carbune, and Abhinav Rastogi. 2023. Rlaif: Scaling reinforcement learning from human feedback with ai feedback. arXiv preprint arXiv:2309.00267 (2023)."},{"key":"e_1_3_2_1_39_1","volume-title":"RLAIF: Scaling Reinforcement Learning from Human Feedback with AI Feedback. CoRR","author":"Lee Harrison","year":"2023","unstructured":"Harrison Lee, Samrat Phatale, Hassan Mansoor, Kellie Lu, Thomas Mesnard, Colton Bishop, Victor Carbune, and Abhinav Rastogi. 2023. RLAIF: Scaling Reinforcement Learning from Human Feedback with AI Feedback. CoRR, Vol. abs\/2309.00267 (2023)."},{"key":"e_1_3_2_1_40_1","volume-title":"Advances in Neural Information Processing Systems","volume":"36","author":"Li Kenneth","year":"2024","unstructured":"Kenneth Li, Oam Patel, Fernanda Vi\u00e9gas, Hanspeter Pfister, and Martin Wattenberg. 2024. Inference-time intervention: Eliciting truthful answers from a language model. Advances in Neural Information Processing Systems, Vol. 36 (2024)."},{"key":"e_1_3_2_1_41_1","volume-title":"DeepInception: Hypnotize Large Language Model to Be Jailbreaker. CoRR","author":"Li Xuan","year":"2023","unstructured":"Xuan Li, Zhanke Zhou, Jianing Zhu, Jiangchao Yao, Tongliang Liu, and Bo Han. 2023. DeepInception: Hypnotize Large Language Model to Be Jailbreaker. CoRR, Vol. abs\/2311.03191 (2023)."},{"key":"e_1_3_2_1_42_1","volume-title":"Ai transparency in the age of llms: A human-centered research roadmap. arXiv preprint arXiv:2306.01941","author":"Vera Liao Q","year":"2023","unstructured":"Q Vera Liao and Jennifer Wortman Vaughan. 2023. Ai transparency in the age of llms: A human-centered research roadmap. arXiv preprint arXiv:2306.01941 (2023)."},{"key":"e_1_3_2_1_43_1","volume-title":"Truthfulqa: Measuring how models mimic human falsehoods. arXiv preprint arXiv:2109.07958","author":"Lin Stephanie","year":"2021","unstructured":"Stephanie Lin, Jacob Hilton, and Owain Evans. 2021. Truthfulqa: Measuring how models mimic human falsehoods. arXiv preprint arXiv:2109.07958 (2021)."},{"key":"e_1_3_2_1_44_1","unstructured":"Fuxiao Liu Tianrui Guan Zongxia Li Lichang Chen Yaser Yacoob Dinesh Manocha and Tianyi Zhou. 2023. Hallusionbench: You see what you think? or you think what you see? an image-context reasoning benchmark challenging for gpt-4v (ision) llava-1.5 and other multi-modality models. arXiv preprint arXiv:2310.14566 (2023)."},{"key":"e_1_3_2_1_45_1","volume-title":"Aligning large multi-modal model with robust instruction tuning. arXiv preprint arXiv:2306.14565","author":"Liu Fuxiao","year":"2023","unstructured":"Fuxiao Liu, Kevin Lin, Linjie Li, Jianfeng Wang, Yaser Yacoob, and Lijuan Wang. 2023. Aligning large multi-modal model with robust instruction tuning. arXiv preprint arXiv:2306.14565 (2023)."},{"key":"e_1_3_2_1_46_1","volume-title":"Proceedings of the 29th International Conference on Computational Linguistics. 378--388","author":"Liu Jiayi","year":"2022","unstructured":"Jiayi Liu, Wei Wei, Zhixuan Chu, Xing Gao, Ji Zhang, Tan Yan, and Yulin Kang. 2022. Incorporating Causal Analysis into Diversified and Logical Response Generation. In Proceedings of the 29th International Conference on Computational Linguistics. 378--388."},{"key":"e_1_3_2_1_47_1","unstructured":"Lei Liu Xiaoyan Yang Junchi Lei Xiaoyang Liu Yue Shen Zhiqiang Zhang Peng Wei Jinjie Gu Zhixuan Chu Zhan Qin et al. 2024. A Survey on Medical Large Language Models: Technology Application Trustworthiness and Future Directions. arXiv preprint arXiv:2406.03712 (2024)."},{"key":"e_1_3_2_1_48_1","volume-title":"AutoDAN: Generating Stealthy Jailbreak Prompts on Aligned Large Language Models. CoRR","author":"Liu Xiaogeng","year":"2023","unstructured":"Xiaogeng Liu, Nan Xu, Muhao Chen, and Chaowei Xiao. 2023. AutoDAN: Generating Stealthy Jailbreak Prompts on Aligned Large Language Models. CoRR, Vol. abs\/2310.04451 (2023)."},{"key":"e_1_3_2_1_49_1","volume-title":"An empirical survey of the effectiveness of debiasing techniques for pre-trained language models. arXiv preprint arXiv:2110.08527","author":"Meade Nicholas","year":"2021","unstructured":"Nicholas Meade, Elinor Poole-Dayan, and Siva Reddy. 2021. An empirical survey of the effectiveness of debiasing techniques for pre-trained language models. arXiv preprint arXiv:2110.08527 (2021)."},{"key":"e_1_3_2_1_50_1","first-page":"17359","article-title":"Locating and editing factual associations in GPT","volume":"35","author":"Meng Kevin","year":"2022","unstructured":"Kevin Meng, David Bau, Alex Andonian, and Yonatan Belinkov. 2022. Locating and editing factual associations in GPT. Advances in Neural Information Processing Systems, Vol. 35 (2022), 17359--17372.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_51_1","volume-title":"Proceedings of the 2013 conference of the north american chapter of the association for computational linguistics: Human language technologies. 746--751","author":"Mikolov Tom\u00e1vs","year":"2013","unstructured":"Tom\u00e1vs Mikolov, Wen-tau Yih, and Geoffrey Zweig. 2013. Linguistic regularities in continuous space word representations. In Proceedings of the 2013 conference of the north american chapter of the association for computational linguistics: Human language technologies. 746--751."},{"key":"e_1_3_2_1_52_1","first-page":"27730","article-title":"Training language models to follow instructions with human feedback","volume":"35","author":"Ouyang Long","year":"2022","unstructured":"Long Ouyang, Jeffrey Wu, Xu Jiang, Diogo Almeida, Carroll Wainwright, Pamela Mishkin, Chong Zhang, Sandhini Agarwal, Katarina Slama, Alex Ray, et al. 2022. Training language models to follow instructions with human feedback. Advances in Neural Information Processing Systems, Vol. 35 (2022), 27730--27744.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_53_1","unstructured":"Nick Pawlowski James Vaughan Joel Jennings and Cheng Zhang. 2023. Answering causal questions with augmented llms. (2023)."},{"key":"e_1_3_2_1_54_1","volume-title":"Causality: models, reasoning and inference","author":"Pearl Judea","unstructured":"Judea Pearl. 2000. Causality: models, reasoning and inference. Vol. 29. Springer."},{"key":"e_1_3_2_1_55_1","volume-title":"The Book of Why","author":"Pearl Judea","unstructured":"Judea Pearl and Dana Mackenzie. 2018. The Book of Why. Basic Books, New York."},{"key":"e_1_3_2_1_56_1","volume-title":"Fine-tuning aligned language models compromises safety, even when users do not intend to! arXiv preprint arXiv:2310.03693","author":"Qi Xiangyu","year":"2023","unstructured":"Xiangyu Qi, Yi Zeng, Tinghao Xie, Pin-Yu Chen, Ruoxi Jia, Prateek Mittal, and Peter Henderson. 2023. Fine-tuning aligned language models compromises safety, even when users do not intend to! arXiv preprint arXiv:2310.03693 (2023)."},{"key":"e_1_3_2_1_57_1","volume-title":"Advances in Neural Information Processing Systems 36: Annual Conference on Neural Information Processing Systems 2023","author":"Rafailov Rafael","year":"2023","unstructured":"Rafael Rafailov, Archit Sharma, Eric Mitchell, Christopher D. Manning, Stefano Ermon, and Chelsea Finn. 2023. Direct Preference Optimization: Your Language Model is Secretly a Reward Model. In Advances in Neural Information Processing Systems 36: Annual Conference on Neural Information Processing Systems 2023, NeurIPS 2023, New Orleans, LA, USA, December 10 - 16, 2023."},{"key":"e_1_3_2_1_58_1","doi-asserted-by":"publisher","DOI":"10.1198\/016214504000001880"},{"key":"e_1_3_2_1_59_1","volume-title":"Evaluating gender bias in machine translation. arXiv preprint arXiv:1906.00591","author":"Stanovsky Gabriel","year":"2019","unstructured":"Gabriel Stanovsky, Noah A Smith, and Luke Zettlemoyer. 2019. Evaluating gender bias in machine translation. arXiv preprint arXiv:1906.00591 (2019)."},{"key":"e_1_3_2_1_60_1","volume-title":"Extracting latent steering vectors from pretrained language models. arXiv preprint arXiv:2205.05124","author":"Subramani Nishant","year":"2022","unstructured":"Nishant Subramani, Nivedita Suresh, and Matthew E Peters. 2022. Extracting latent steering vectors from pretrained language models. arXiv preprint arXiv:2205.05124 (2022)."},{"key":"e_1_3_2_1_61_1","volume-title":"Large Language Models for Data Annotation: A Survey. arXiv preprint arXiv:2402.13446","author":"Tan Zhen","year":"2024","unstructured":"Zhen Tan, Alimohammad Beigi, Song Wang, Ruocheng Guo, Amrita Bhattacharjee, Bohan Jiang, Mansooreh Karami, Jundong Li, Lu Cheng, and Huan Liu. 2024. Large Language Models for Data Annotation: A Survey. arXiv preprint arXiv:2402.13446 (2024)."},{"key":"e_1_3_2_1_62_1","volume-title":"Deception and Manipulation in Generative AI. arXiv preprint arXiv:2401.11335","author":"Tarsney Christian","year":"2024","unstructured":"Christian Tarsney. 2024. Deception and Manipulation in Generative AI. arXiv preprint arXiv:2401.11335 (2024)."},{"key":"e_1_3_2_1_63_1","volume-title":"BERT rediscovers the classical NLP pipeline. arXiv preprint arXiv:1905.05950","author":"Tenney Ian","year":"2019","unstructured":"Ian Tenney, Dipanjan Das, and Ellie Pavlick. 2019. BERT rediscovers the classical NLP pipeline. arXiv preprint arXiv:1905.05950 (2019)."},{"key":"e_1_3_2_1_64_1","unstructured":"Hugo Touvron Louis Martin Kevin Stone Peter Albert Amjad Almahairi Yasmine Babaei Nikolay Bashlykov Soumya Batra Prajjwal Bhargava Shruti Bhosale et al. 2023. Llama 2: Open foundation and fine-tuned chat models. arXiv preprint arXiv:2307.09288 (2023)."},{"key":"e_1_3_2_1_65_1","volume-title":"Activation Addition: Steering Language Models Without Optimization. arXiv preprint arXiv:2308.10248","author":"Turner Alex","year":"2023","unstructured":"Alex Turner, Lisa Thiergart, David Udell, Gavin Leech, Ulisse Mini, and Monte MacDiarmid. 2023. Activation Addition: Steering Language Models Without Optimization. arXiv preprint arXiv:2308.10248 (2023)."},{"key":"e_1_3_2_1_66_1","volume-title":"Bridging Causal Discovery and Large Language Models: A Comprehensive Survey of Integrative Approaches and Future Directions. arXiv preprint arXiv:2402.11068","author":"Wan Guangya","year":"2024","unstructured":"Guangya Wan, Yuqi Wu, Mengxuan Hu, Zhixuan Chu, and Sheng Li. 2024. Bridging Causal Discovery and Large Language Models: A Comprehensive Survey of Integrative Approaches and Future Directions. arXiv preprint arXiv:2402.11068 (2024)."},{"key":"e_1_3_2_1_67_1","volume-title":"Backdoor activation attack: Attack large language models using activation steering for safety-alignment. arXiv preprint arXiv:2311.09433","author":"Wang Haoran","year":"2023","unstructured":"Haoran Wang and Kai Shu. 2023. Backdoor activation attack: Attack large language models using activation steering for safety-alignment. arXiv preprint arXiv:2311.09433 (2023)."},{"key":"e_1_3_2_1_68_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v38i17.29887"},{"key":"e_1_3_2_1_69_1","volume-title":"Kiran Ramnath, Sougata Chaudhuri, Shubham Mehrotra, Xiang-Bo Mao, Sitaram Asur, et al.","author":"Wang Zhichao","year":"2024","unstructured":"Zhichao Wang, Bin Bi, Shiva Kumar Pentyala, Kiran Ramnath, Sougata Chaudhuri, Shubham Mehrotra, Xiang-Bo Mao, Sitaram Asur, et al. 2024. A Comprehensive Survey of LLM Alignment Techniques: RLHF, RLAIF, PPO, DPO and More. arXiv preprint arXiv:2407.16216 (2024)."},{"key":"e_1_3_2_1_70_1","volume-title":"Jailbreak and Guard Aligned Language Models with Only Few In-Context Demonstrations. CoRR","author":"Wei Zeming","year":"2023","unstructured":"Zeming Wei, Yifei Wang, and Yisen Wang. 2023. Jailbreak and Guard Aligned Language Models with Only Few In-Context Demonstrations. CoRR, Vol. abs\/2310.06387 (2023)."},{"key":"e_1_3_2_1_71_1","volume-title":"Unveiling the implicit toxicity in large language models. arXiv preprint arXiv:2311.17391","author":"Wen Jiaxin","year":"2023","unstructured":"Jiaxin Wen, Pei Ke, Hao Sun, Zhexin Zhang, Chengfei Li, Jinfeng Bai, and Minlie Huang. 2023. Unveiling the implicit toxicity in large language models. arXiv preprint arXiv:2311.17391 (2023)."},{"key":"e_1_3_2_1_72_1","volume-title":"A survey on causal inference. ACM Transactions on Knowledge Discovery from Data (TKDD)","author":"Yao Liuyi","year":"2021","unstructured":"Liuyi Yao, Zhixuan Chu, Sheng Li, Yaliang Li, Jing Gao, and Aidong Zhang. 2021. A survey on causal inference. ACM Transactions on Knowledge Discovery from Data (TKDD), Vol. 15, 5 (2021), 1--46."},{"key":"e_1_3_2_1_73_1","volume-title":"International Conference on Machine Learning. PMLR, 26750--26771","author":"Zhang Jiayao","year":"2022","unstructured":"Jiayao Zhang, Hongming Zhang, Weijie Su, and Dan Roth. 2022. Rock: Causal inference principles for reasoning about commonsense causality. In International Conference on Machine Learning. PMLR, 26750--26771."},{"key":"e_1_3_2_1_74_1","doi-asserted-by":"publisher","DOI":"10.1145\/3658673"},{"key":"e_1_3_2_1_75_1","volume-title":"Mquake: Assessing knowledge editing in language models via multi-hop questions. arXiv preprint arXiv:2305.14795","author":"Zhong Zexuan","year":"2023","unstructured":"Zexuan Zhong, Zhengxuan Wu, Christopher D Manning, Christopher Potts, and Danqi Chen. 2023. Mquake: Assessing knowledge editing in language models via multi-hop questions. arXiv preprint arXiv:2305.14795 (2023)."},{"key":"e_1_3_2_1_76_1","unstructured":"Andy Zou Long Phan Sarah Chen James Campbell Phillip Guo Richard Ren Alexander Pan Xuwang Yin Mantas Mazeika Ann-Kathrin Dombrowski et al. 2023. Representation engineering: A top-down approach to ai transparency. arXiv preprint arXiv:2310.01405 (2023)."},{"key":"e_1_3_2_1_77_1","volume-title":"Universal and transferable adversarial attacks on aligned language models. arXiv preprint arXiv:2307.15043","author":"Zou Andy","year":"2023","unstructured":"Andy Zou, Zifan Wang, J Zico Kolter, and Matt Fredrikson. 2023. Universal and transferable adversarial attacks on aligned language models. arXiv preprint arXiv:2307.15043 (2023)."}],"event":{"name":"CCS '24: ACM SIGSAC Conference on Computer and Communications Security","location":"Salt Lake City UT USA","acronym":"CCS '24","sponsor":["SIGSAC ACM Special Interest Group on Security, Audit, and Control"]},"container-title":["Proceedings of the 2024 on ACM SIGSAC Conference on Computer and Communications Security"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3658644.3690217","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3658644.3690217","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,22]],"date-time":"2025-08-22T06:11:44Z","timestamp":1755843104000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3658644.3690217"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,12,2]]},"references-count":77,"alternative-id":["10.1145\/3658644.3690217","10.1145\/3658644"],"URL":"https:\/\/doi.org\/10.1145\/3658644.3690217","relation":{},"subject":[],"published":{"date-parts":[[2024,12,2]]},"assertion":[{"value":"2024-12-09","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}