{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,2]],"date-time":"2026-07-02T16:17:53Z","timestamp":1783009073290,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":97,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,12,2]],"date-time":"2024-12-02T00:00:00Z","timestamp":1733097600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,12,2]]},"DOI":"10.1145\/3658644.3690322","type":"proceedings-article","created":{"date-parts":[[2024,12,9]],"date-time":"2024-12-09T12:19:20Z","timestamp":1733746760000},"page":"1151-1165","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":8,"title":["<scp>Legilimens:<\/scp>\n            Practical and Unified Content Moderation for Large Language Model Services"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0009-9659-5259","authenticated-orcid":false,"given":"Jialin","family":"Wu","sequence":"first","affiliation":[{"name":"Zhejiang University, Hangzhou, Zhejiang, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8262-7813","authenticated-orcid":false,"given":"Jiangyi","family":"Deng","sequence":"additional","affiliation":[{"name":"Zhejiang University, Hangzhou, Zhejiang, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-7945-3987","authenticated-orcid":false,"given":"Shengyuan","family":"Pang","sequence":"additional","affiliation":[{"name":"Zhejiang University, Hangzhou, Zhejiang, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1382-0679","authenticated-orcid":false,"given":"Yanjiao","family":"Chen","sequence":"additional","affiliation":[{"name":"Zhejiang University, Hangzhou, Zhejiang, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-2557-2438","authenticated-orcid":false,"given":"Jiayang","family":"Xu","sequence":"additional","affiliation":[{"name":"Zhejiang University, Hangzhou, Zhejiang, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9686-4369","authenticated-orcid":false,"given":"Xinfeng","family":"Li","sequence":"additional","affiliation":[{"name":"Zhejiang University, Hangzhou, Zhejiang, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5043-9148","authenticated-orcid":false,"given":"Wenyuan","family":"Xu","sequence":"additional","affiliation":[{"name":"Zhejiang University, Hangzhou, Zhejiang, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2024,12,9]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"A General Language Assistant as a Laboratory for Alignment. arXiv preprint arXiv: 2112.00861","author":"Askell Amanda","year":"2021","unstructured":"Amanda Askell, Yuntao Bai, Anna Chen, Dawn Drain, Deep Ganguli, Tom Henighan, Andy Jones, Nicholas Joseph, Benjamin Mann, Nova DasSarma, Nelson Elhage, Zac Hatfield-Dodds, Danny Hernandez, Jackson Kernion, Kamal Ndousse, Catherine Olsson, Dario Amodei, Tom B. Brown, Jack Clark, Sam McCandlish, Chris Olah, and Jared Kaplan. 2021. A General Language Assistant as a Laboratory for Alignment. arXiv preprint arXiv: 2112.00861 (2021)."},{"key":"e_1_3_2_1_2_1","volume-title":"The Internal State of an LLM Knows When It's Lying. arXiv preprint arXiv:2304.13734","author":"Azaria Amos","year":"2023","unstructured":"Amos Azaria and Tom Mitchell. 2023. The Internal State of an LLM Knows When It's Lying. arXiv preprint arXiv:2304.13734 (2023)."},{"key":"e_1_3_2_1_3_1","volume-title":"Jamie Ryan Kiros, and Geoffrey E. Hinton","author":"Ba Lei Jimmy","year":"2016","unstructured":"Lei Jimmy Ba, Jamie Ryan Kiros, and Geoffrey E. Hinton. 2016. Layer Normalization. arXiv preprint arXiv: 1607.06450 (2016)."},{"key":"e_1_3_2_1_4_1","volume-title":"Sheer El Showk","author":"Bai Yuntao","unstructured":"Yuntao Bai, Andy Jones, Kamal Ndousse, Amanda Askell, Anna Chen, Nova DasSarma, Dawn Drain, Stanislav Fort, Deep Ganguli, Tom Henighan, Nicholas Joseph, Saurav Kadavath, Jackson Kernion, Tom Conerly, Sheer El Showk, Nelson Elhage, Zac Hatfield-Dodds, Danny Hernandez, Tristan Hume, Scott Johnston, Shauna Kravec, Liane Lovitt, Neel Nanda, Catherine Olsson, Dario Amodei, Tom B. Brown, Jack Clark, Sam McCandlish, Chris Olah, Benjamin Mann, and Jared Kaplan. 2022. Training a Helpful and Harmless Assistant with Reinforcement Learning from Human Feedback. arXiv preprint arXiv: 2204.0586 (2022)."},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1561\/3300000041"},{"key":"e_1_3_2_1_6_1","volume-title":"Safety-Tuned Llamas: Lessons from Improving the Safety of Large Language Models that Follow Instructions. arXiv preprint arXiv: 2309.07875","author":"Bianchi Federico","year":"2023","unstructured":"Federico Bianchi, Mirac Suzgun, Giuseppe Attanasio, Paul R\u00f6ttger, Dan Jurafsky, Tatsunori Hashimoto, and James Zou. 2023. Safety-Tuned Llamas: Lessons from Improving the Safety of Large Language Models that Follow Instructions. arXiv preprint arXiv: 2309.07875 (2023)."},{"key":"e_1_3_2_1_7_1","volume-title":"International Conference on Machine Learning. PMLR.","author":"Biderman Stella","year":"2023","unstructured":"Stella Biderman, Hailey Schoelkopf, Quentin Gregory Anthony, Herbie Bradley, Kyle O'Brien, Eric Hallahan, Mohammad Aflah Khan, Shivanshu Purohit, USVSN Sai Prashanth, Edward Raff, et al. 2023. Pythia: A Suite for Analyzing Large Language Models across Training and Scaling. In International Conference on Machine Learning. PMLR."},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i05.6239"},{"key":"e_1_3_2_1_9_1","volume-title":"Conference on Neural Information Processing Systems. PMLR.","author":"Brown Tom","year":"2020","unstructured":"Tom Brown, Benjamin Mann, Nick Ryder, Melanie Subbiah, Jared D Kaplan, Prafulla Dhariwal, Arvind Neelakantan, Pranav Shyam, Girish Sastry, Amanda Askell, et al. 2020. Language Models are Few-shot Learners. In Conference on Neural Information Processing Systems. PMLR."},{"key":"e_1_3_2_1_10_1","unstructured":"Tom Brown Benjamin Mann Nick Ryder Melanie Subbiah Jared D Kaplan Prafulla Dhariwal Arvind Neelakantan Pranav Shyam Girish Sastry Amanda Askell et al. 2020. Language models are few-shot learners. Advances in neural information processing systems Vol. 33 (2020) 1877--1901."},{"key":"e_1_3_2_1_11_1","volume-title":"Michelle L. Mazurek, Katie Shilton, and Hal Daum\u00e9 III.","author":"Cao Yang Trista","year":"2023","unstructured":"Yang Trista Cao, Lovely-Frances Domingo, Sarah Ann Gilbert, Michelle L. Mazurek, Katie Shilton, and Hal Daum\u00e9 III. 2023. Toxicity Detection is NOT All You Need: Measuring the Gaps to Supporting Volunteer Content Moderators. arXiv preprint arXiv: 2311.07879 (2023)."},{"key":"e_1_3_2_1_12_1","volume-title":"Jailbreaking Black Box Large Language Models in Twenty Queries. arXiv preprint arXiv: 2310.08419","author":"Chao Patrick","year":"2023","unstructured":"Patrick Chao, Alexander Robey, Edgar Dobriban, Hamed Hassani, George J Pappas, and Eric Wong. 2023. Jailbreaking Black Box Large Language Models in Twenty Queries. arXiv preprint arXiv: 2310.08419 (2023)."},{"key":"e_1_3_2_1_13_1","volume-title":"INSIDE: LLMs' Internal States Retain the Power of Hallucination Detection. arXiv preprint arXiv:2402.03744","author":"Chen Chao","year":"2024","unstructured":"Chao Chen, Kai Liu, Ze Chen, Yi Gu, Yue Wu, Mingyuan Tao, Zhihang Fu, and Jieping Ye. 2024. INSIDE: LLMs' Internal States Retain the Power of Hallucination Detection. arXiv preprint arXiv:2402.03744 (2024)."},{"key":"e_1_3_2_1_14_1","volume-title":"Generating Long Sequences with Sparse Transformers. arXiv preprint arXiv","author":"Child Rewon","year":"1904","unstructured":"Rewon Child, Scott Gray, Alec Radford, and Ilya Sutskever. 2019. Generating Long Sequences with Sparse Transformers. arXiv preprint arXiv: 1904.10509 (2019)."},{"key":"e_1_3_2_1_15_1","volume-title":"Comprehensive Assessment of Jailbreak Attacks Against LLMs. arXiv preprint arXiv: 2402.05668","author":"Chu Junjie","year":"2024","unstructured":"Junjie Chu, Yugeng Liu, Ziqing Yang, Xinyue Shen, Michael Backes, and Yang Zhang. 2024. Comprehensive Assessment of Jailbreak Attacks Against LLMs. arXiv preprint arXiv: 2402.05668 (2024)."},{"key":"e_1_3_2_1_16_1","unstructured":"Hyung Won Chung Le Hou Shayne Longpre Barret Zoph Yi Tay William Fedus Eric Li Xuezhi Wang Mostafa Dehghani Siddhartha Brahma Albert Webson Shixiang Shane Gu Zhuyun Dai Mirac Suzgun Xinyun Chen Aakanksha Chowdhery Sharan Narang Gaurav Mishra Adams Yu Vincent Y. Zhao Yanping Huang Andrew M. Dai Hongkun Yu Slav Petrov Ed H. Chi Jeff Dean Jacob Devlin Adam Roberts Denny Zhou Quoc V. Le and Jason Wei. 2022. Scaling Instruction-Finetuned Language Models. arXiv preprint arXiv: 2210.11416 (2022)."},{"key":"e_1_3_2_1_17_1","volume-title":"Hello Dolly: Democratizing the Magic of ChatGPT with Open Models. https:\/\/www.databricks.com\/blog\/2023\/03\/24\/hello-dolly-democratizing-magic-chatgpt-open-models.html.","author":"Conover Mike","year":"2023","unstructured":"Mike Conover, Matt Hayes, Ankit Mathur, Xiangrui Meng, Jianwei Xie, Jun Wan, Ali Ghodsi, Patrick Wendell, and Matei Zaharia. 2023. Hello Dolly: Democratizing the Magic of ChatGPT with Open Models. https:\/\/www.databricks.com\/blog\/2023\/03\/24\/hello-dolly-democratizing-magic-chatgpt-open-models.html."},{"key":"e_1_3_2_1_18_1","volume-title":"Free Dolly: Introducing the World's First Truly Open Instruction-Tuned LLM. https:\/\/www.databricks.com\/blog\/2023\/04\/12\/dolly-first-open-commercially-viable-instruction-tuned-llm.","author":"Conover Mike","year":"2023","unstructured":"Mike Conover, Matt Hayes, Ankit Mathur, Jianwei Xie, Jun Wan, Sam Shah, Ali Ghodsi, Patrick Wendell, Matei Zaharia, and Reynold Xin. 2023. Free Dolly: Introducing the World's First Truly Open Instruction-Tuned LLM. https:\/\/www.databricks.com\/blog\/2023\/04\/12\/dolly-first-open-commercially-viable-instruction-tuned-llm."},{"key":"e_1_3_2_1_19_1","volume-title":"arXiv preprint arXiv: 2401.05778","author":"Cui Tianyu","year":"2024","unstructured":"Tianyu Cui, Yanling Wang, Chuanpu Fu, Yong Xiao, Sijia Li, Xinhao Deng, Yunpeng Liu, Qinglin Zhang, Ziyi Qiu, Peiyang Li, Zhixing Tan, Junwu Xiong, Xinyu Kong, Zujie Wen, Ke Xu, and Qi Li. 2024. Risk Taxonomy, Mitigation, and Assessment Benchmarks of Large Language Model Systems. arXiv preprint arXiv: 2401.05778 (2024)."},{"key":"e_1_3_2_1_20_1","volume-title":"Language Modeling with Gated Convolutional Networks. In International conference on machine learning. PMLR.","author":"Dauphin Yann N","year":"2017","unstructured":"Yann N Dauphin, Angela Fan, Michael Auli, and David Grangier. 2017. Language Modeling with Gated Convolutional Networks. In International conference on machine learning. PMLR."},{"key":"e_1_3_2_1_21_1","volume-title":"Automated Hate Speech Detection and the Problem of Offensive Language. In International Conference on Web and Social Media. AAAI.","author":"Davidson Thomas","year":"2017","unstructured":"Thomas Davidson, Dana Warmsley, Michael W. Macy, and Ingmar Weber. 2017. Automated Hate Speech Detection and the Problem of Offensive Language. In International Conference on Web and Social Media. AAAI."},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D19-1461"},{"key":"e_1_3_2_1_23_1","volume-title":"Unified Language Model Pre-Training for Natural Language Understanding and Generation. In Conference on Neural Information Processing Systems. PMLR.","author":"Dong Li","year":"2019","unstructured":"Li Dong, Nan Yang, Wenhui Wang, Furu Wei, Xiaodong Liu, Yu Wang, Jianfeng Gao, Ming Zhou, and Hsiao-Wuen Hon. 2019. Unified Language Model Pre-Training for Natural Language Understanding and Generation. In Conference on Neural Information Processing Systems. PMLR."},{"key":"e_1_3_2_1_24_1","volume-title":"GLM: General Language Model Pretraining with Autoregressive Blank Infilling. In Annual Meeting of the Association for Computational Linguistics.","author":"Du Zhengxiao","year":"2022","unstructured":"Zhengxiao Du, Yujie Qian, Xiao Liu, Ming Ding, Jiezhong Qiu, Zhilin Yang, and Jie Tang. 2022. GLM: General Language Model Pretraining with Autoregressive Blank Infilling. In Annual Meeting of the Association for Computational Linguistics."},{"key":"e_1_3_2_1_25_1","volume-title":"Andreas R\u00fcckl\u00e9, Ji-Ung Lee, Claudia Schulz, Mohsen Mesgar, Krishnkant Swarnkar, Edwin Simpson, and Iryna Gurevych.","author":"Eger Steffen","year":"2019","unstructured":"Steffen Eger, G\u00f6zde G\u00fcl Sahin, Andreas R\u00fcckl\u00e9, Ji-Ung Lee, Claudia Schulz, Mohsen Mesgar, Krishnkant Swarnkar, Edwin Simpson, and Iryna Gurevych. 2019. Text Processing Like Humans Do: Visually Attacking and Shielding NLP Systems. In Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies. Association for Computational Linguistics."},{"key":"e_1_3_2_1_26_1","volume-title":"ACM Comput. Surv.","volume":"51","author":"Fortuna Paula","year":"2018","unstructured":"Paula Fortuna and S\u00e9rgio Nunes. 2018. A Survey on Automatic Detection of Hate Speech in Text. ACM Comput. Surv., Vol. 51, 4 (2018), 85:1--85:30."},{"key":"e_1_3_2_1_27_1","volume-title":"Reduce Harms: Methods, Scaling Behaviors, and Lessons Learned. arXiv preprint arXiv: 2209.07858","author":"Ganguli Deep","year":"2022","unstructured":"Deep Ganguli, Liane Lovitt, Jackson Kernion, Amanda Askell, Yuntao Bai, Saurav Kadavath, Ben Mann, Ethan Perez, Nicholas Schiefer, Kamal Ndousse, et al. 2022. Red Teaming Language Models to Reduce Harms: Methods, Scaling Behaviors, and Lessons Learned. arXiv preprint arXiv: 2209.07858 (2022)."},{"key":"e_1_3_2_1_28_1","unstructured":"Google Jigsaw. 2017. Jigsaw Unintended Bias in Toxicity Classification. https:\/\/www.kaggle.com\/c\/jigsaw-unintended-bias-in-toxicity-classification."},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1177\/2053951719897945"},{"key":"e_1_3_2_1_30_1","unstructured":"Andrew Griffin. 2023. ChatGPT Plus: OpenAI Stops Premium Signups after Major Update. https:\/\/www.independent.co.uk\/tech\/chatgpt-plus-free-premium-paid-subscription-sign-up-b2447941.html."},{"key":"e_1_3_2_1_31_1","volume-title":"Cold-attack: Jailbreaking llms with stealthiness and controllability. arXiv preprint arXiv:2402.08679","author":"Guo Xingang","year":"2024","unstructured":"Xingang Guo, Fangxu Yu, Huan Zhang, Lianhui Qin, and Bin Hu. 2024. Cold-attack: Jailbreaking llms with stealthiness and controllability. arXiv preprint arXiv:2402.08679 (2024)."},{"key":"e_1_3_2_1_32_1","volume-title":"Deep Residual Learning for Image Recognition. In IEEE Conference on Computer Vision and Pattern Recognition. IEEE Computer Society.","author":"He Kaiming","year":"2016","unstructured":"Kaiming He, Xiangyu Zhang, Shaoqing Ren, and Jian Sun. 2016. Deep Residual Learning for Image Recognition. In IEEE Conference on Computer Vision and Pattern Recognition. IEEE Computer Society."},{"key":"e_1_3_2_1_33_1","volume-title":"Harnessing the Power of ChatGPT in Fake News: An In-Depth Exploration in Generation, Detection and Explanation. arXiv preprint arXiv: 2310.05046","author":"Huang Yue","year":"2023","unstructured":"Yue Huang and Lichao Sun. 2023. Harnessing the Power of ChatGPT in Fake News: An In-Depth Exploration in Generation, Detection and Explanation. arXiv preprint arXiv: 2310.05046 (2023)."},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.2312.06674"},{"key":"e_1_3_2_1_35_1","volume-title":"Adversarial Example Generation with Syntactically Controlled Paraphrase Networks. In Annual Conference of the North American Chapter of the Association for Computational Linguistics.","author":"Iyyer Mohit","year":"2018","unstructured":"Mohit Iyyer, John Wieting, Kevin Gimpel, and Luke Zettlemoyer. 2018. Adversarial Example Generation with Syntactically Controlled Paraphrase Networks. In Annual Conference of the North American Chapter of the Association for Computational Linguistics."},{"key":"e_1_3_2_1_36_1","volume-title":"Conference on Neural Information Processing Systems. PMLR.","author":"Ji Jiaming","year":"2023","unstructured":"Jiaming Ji, Mickel Liu, Josef Dai, Xuehai Pan, Chi Zhang, Ce Bian, Boyuan Chen, Ruiyang Sun, Yizhou Wang, and Yaodong Yang. 2023. BeaverTails: Towards Improved Safety Alignment of LLM via a Human-Preference Dataset. In Conference on Neural Information Processing Systems. PMLR."},{"key":"e_1_3_2_1_37_1","unstructured":"Google Jigsaw. 2017. Perspective API. https:\/\/www.perspectiveapi.com\/."},{"key":"e_1_3_2_1_38_1","volume-title":"Constructing Interval Variables via Faceted Rasch Measurement and Multitask Deep Learning: a Hate Speech Application. arXiv preprint arXiv","author":"Kennedy Chris J","year":"2009","unstructured":"Chris J Kennedy, Geoff Bacon, Alexander Sahn, and Claudia von Vacano. 2020. Constructing Interval Variables via Faceted Rasch Measurement and Multitask Deep Learning: a Hate Speech Application. arXiv preprint arXiv: 2009.10277 (2020)."},{"key":"e_1_3_2_1_39_1","volume-title":"Harris","author":"Kim Jinhwa","year":"2023","unstructured":"Jinhwa Kim, Ali Derakhshan, and Ian G. Harris. 2023. Robust Safety Classifier for Large Language Models: Adversarial Prompt Shield. arXiv preprint arXiv: 2311.00172 (2023)."},{"key":"e_1_3_2_1_40_1","volume-title":"Meet DAN -- The `JAILBREAK' Version of ChatGPT and How to Use it -- AI Unchained and Unfiltered. https:\/\/platform.openai.com\/docs\/guides\/moderation.","author":"King Michael","year":"2023","unstructured":"Michael King. 2023. Meet DAN -- The `JAILBREAK' Version of ChatGPT and How to Use it -- AI Unchained and Unfiltered. https:\/\/platform.openai.com\/docs\/guides\/moderation."},{"key":"e_1_3_2_1_41_1","volume-title":"Adam: A Method for Stochastic Optimization. In International Conference on Learning Representations. OpenReview.net.","author":"Diederik","unstructured":"Diederik P. Kingma and Jimmy Ba. 2015. Adam: A Method for Stochastic Optimization. In International Conference on Learning Representations. OpenReview.net."},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v27i1.8539"},{"key":"e_1_3_2_1_43_1","volume-title":"Findings of the Association for Computational Linguistics: EMNLP","author":"Li Haoran","unstructured":"Haoran Li, Dadi Guo, Wei Fan, Mingshi Xu, Jie Huang, Fanpu Meng, and Yangqiu Song. 2023. Multi-Step Jailbreaking Privacy Attacks on ChatGPT. In Findings of the Association for Computational Linguistics: EMNLP. Association for Computational Linguistics."},{"key":"e_1_3_2_1_44_1","volume-title":"A Survey of Transformers. arXiv preprint arXiv: 2106.04554","author":"Lin Tianyang","year":"2021","unstructured":"Tianyang Lin, Yuxin Wang, Xiangyang Liu, and Xipeng Qiu. 2021. A Survey of Transformers. arXiv preprint arXiv: 2106.04554 (2021)."},{"key":"e_1_3_2_1_45_1","volume-title":"Jailbreaking ChatGPT via Prompt Engineering: An Empirical Study. arXiv preprint arXiv: 2305.13860","author":"Liu Yi","year":"2023","unstructured":"Yi Liu, Gelei Deng, Zhengzi Xu, Yuekang Li, Yaowen Zheng, Ying Zhang, Lida Zhao, Tianwei Zhang, and Yang Liu. 2023. Jailbreaking ChatGPT via Prompt Engineering: An Empirical Study. arXiv preprint arXiv: 2305.13860 (2023)."},{"key":"e_1_3_2_1_46_1","volume-title":"Adapting Large Language Models for Content Moderation: Pitfalls in Data Engineering and Supervised Fine-Tuning. arXiv preprint arXiv: 2310.03400","author":"Ma Huan","year":"2023","unstructured":"Huan Ma, Changqing Zhang, Huazhu Fu, Peilin Zhao, and Bingzhe Wu. 2023. Adapting Large Language Models for Content Moderation: Pitfalls in Data Engineering and Supervised Fine-Tuning. arXiv preprint arXiv: 2310.03400 (2023)."},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v37i12.26752"},{"key":"e_1_3_2_1_48_1","volume-title":"HateXplain: A Benchmark Dataset for Explainable Hate Speech Detection. In AAAI Conference on Artificial Intelligence.","author":"Mathew Binny","year":"2021","unstructured":"Binny Mathew, Punyajoy Saha, Seid Muhie Yimam, Chris Biemann, Pawan Goyal, and Animesh Mukherjee. 2021. HateXplain: A Benchmark Dataset for Explainable Hate Speech Detection. In AAAI Conference on Artificial Intelligence."},{"key":"e_1_3_2_1_49_1","unstructured":"Mantas Mazeika Long Phan Xuwang Yin Andy Zou Zifan Wang Norman Mu Elham Sakhaee Nathaniel Li Steven Basart Bo Li et al. 2024. HarmBench: A Standardized Evaluation Framework for Automated Red Teaming and Robust Refusal. arXiv preprint arXiv: 2402.04249 (2024)."},{"key":"e_1_3_2_1_50_1","volume-title":"Tree of Attacks: Jailbreaking Black-Box LLMs Automatically. arXiv preprint arXiv: 2312.02119","author":"Mehrotra Anay","year":"2023","unstructured":"Anay Mehrotra, Manolis Zampetakis, Paul Kassianik, Blaine Nelson, Hyrum Anderson, Yaron Singer, and Amin Karbasi. 2023. Tree of Attacks: Jailbreaking Black-Box LLMs Automatically. arXiv preprint arXiv: 2312.02119 (2023)."},{"key":"e_1_3_2_1_51_1","volume-title":"Analyzing Norm Violations in Live-Stream Chat. In Conference on Empirical Methods in Natural Language Processing. Association for Computational Linguistics.","author":"Moon Jihyung","year":"2023","unstructured":"Jihyung Moon, Dong-Ho Lee, Hyundong Cho, Woojeong Jin, Chan Young Park, Minwoo Kim, Jonathan May, Jay Pujara, and Sungjoon Park. 2023. Analyzing Norm Violations in Live-Stream Chat. In Conference on Empirical Methods in Natural Language Processing. Association for Computational Linguistics."},{"key":"e_1_3_2_1_52_1","volume-title":"Abubakar Yakubu Zandam, and Isa Inuwa-Dutse","author":"Muhammad Fatima Adam","year":"2023","unstructured":"Fatima Adam Muhammad, Abubakar Yakubu Zandam, and Isa Inuwa-Dutse. 2023. Detection of Offensive and Threatening Online Content in a Low Resource Language. arXiv preprint arXiv: 2311.10541 (2023)."},{"key":"e_1_3_2_1_53_1","unstructured":"Huu Nguyen Sameer Suri Ken Tsui Shahules786 Together.xyz team and Christoph Schuhmann. 2023. The OIG Dataset. https:\/\/laion.ai\/blog\/oig-dataset\/."},{"key":"e_1_3_2_1_54_1","volume-title":"Abusive Language Detection in Online User Content. In International Conference on World Wide Web. ACM.","author":"Nobata Chikashi","year":"2016","unstructured":"Chikashi Nobata, Joel R. Tetreault, Achint Thomas, Yashar Mehdad, and Yi Chang. 2016. Abusive Language Detection in Online User Content. In International Conference on World Wide Web. ACM."},{"key":"e_1_3_2_1_55_1","unstructured":"OpenAI. 2023. GPT-4 Technical Report. arXiv preprint arXiv: 2303.08774 (2023)."},{"key":"e_1_3_2_1_56_1","unstructured":"OpenAI. 2023. Moderation. https:\/\/platform.openai.com\/docs\/guides\/moderation."},{"key":"e_1_3_2_1_57_1","volume-title":"Conference on Neural Information Processing Systems. PMLR.","author":"Ouyang Long","year":"2022","unstructured":"Long Ouyang, Jeffrey Wu, Xu Jiang, Diogo Almeida, Carroll L. Wainwright, Pamela Mishkin, Chong Zhang, Sandhini Agarwal, Katarina Slama, Alex Ray, John Schulman, Jacob Hilton, Fraser Kelton, Luke Miller, Maddie Simens, Amanda Askell, Peter Welinder, Paul F. Christiano, Jan Leike, and Ryan Lowe. 2022. Training Language Models to Follow Instructions with Human Feedback. In Conference on Neural Information Processing Systems. PMLR."},{"key":"e_1_3_2_1_58_1","volume-title":"High-Performance Deep Learning Library. In Conference on Neural Information Processing Systems. PMLR.","author":"Paszke Adam","year":"2019","unstructured":"Adam Paszke, Sam Gross, Francisco Massa, Adam Lerer, James Bradbury, Gregory Chanan, Trevor Killeen, Zeming Lin, Natalia Gimelshein, Luca Antiga, Alban Desmaison, Andreas K\u00f6pf, Edward Z. Yang, Zachary DeVito, Martin Raison, Alykhan Tejani, Sasank Chilamkurthy, Benoit Steiner, Lu Fang, Junjie Bai, and Soumith Chintala. 2019. PyTorch: An Imperative Style, High-Performance Deep Learning Library. In Conference on Neural Information Processing Systems. PMLR."},{"key":"e_1_3_2_1_59_1","volume-title":"Advprompter: Fast adaptive adversarial prompting for llms. arXiv preprint arXiv:2404.16873","author":"Paulus Anselm","year":"2024","unstructured":"Anselm Paulus, Arman Zharmagambetov, Chuan Guo, Brandon Amos, and Yuandong Tian. 2024. Advprompter: Fast adaptive adversarial prompting for llms. arXiv preprint arXiv:2404.16873 (2024)."},{"key":"e_1_3_2_1_60_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.acl-main.396"},{"key":"e_1_3_2_1_61_1","volume-title":"The RefinedWeb Dataset for Falcon LLM: Outperforming Curated Corpora with Web Data, and Web Data Only. arXiv preprint arXiv: 2306.01116","author":"Penedo Guilherme","year":"2023","unstructured":"Guilherme Penedo, Quentin Malartic, Daniel Hesslow, Ruxandra Cojocaru, Alessandro Cappelli, Hamza Alobeidli, Baptiste Pannier, Ebtesam Almazrouei, and Julien Launay. 2023. The RefinedWeb Dataset for Falcon LLM: Outperforming Curated Corpora with Web Data, and Web Data Only. arXiv preprint arXiv: 2306.01116 (2023)."},{"key":"e_1_3_2_1_62_1","volume-title":"Instruction Tuning with GPT-4. arXiv preprint arXiv: 2304.03277","author":"Peng Baolin","year":"2023","unstructured":"Baolin Peng, Chunyuan Li, Pengcheng He, Michel Galley, and Jianfeng Gao. 2023. Instruction Tuning with GPT-4. arXiv preprint arXiv: 2304.03277 (2023)."},{"key":"e_1_3_2_1_63_1","volume-title":"Even When Users do not Intend to! arXiv preprint arXiv: 2310.03693","author":"Qi Xiangyu","year":"2023","unstructured":"Xiangyu Qi, Yi Zeng, Tinghao Xie, Pin-Yu Chen, Ruoxi Jia, Prateek Mittal, and Peter Henderson. 2023. Fine-Tuning Aligned Language Models Compromises Safety, Even When Users do not Intend to! arXiv preprint arXiv: 2310.03693 (2023)."},{"key":"e_1_3_2_1_64_1","unstructured":"Jack W. Rae Sebastian Borgeaud Trevor Cai Katie Millican Jordan Hoffmann H. Francis Song John Aslanides Sarah Henderson Roman Ring Susannah Young Eliza Rutherford Tom Hennigan Jacob Menick Albin Cassirer Richard Powell George van den Driessche Lisa Anne Hendricks Maribeth Rauh Po-Sen Huang Amelia Glaese Johannes Welbl Sumanth Dathathri Saffron Huang Jonathan Uesato John Mellor Irina Higgins Antonia Creswell Nat McAleese Amy Wu Erich Elsen Siddhant M. Jayakumar Elena Buchatskaya David Budden Esme Sutherland Karen Simonyan Michela Paganini Laurent Sifre Lena Martens Xiang Lorraine Li Adhiguna Kuncoro Aida Nematzadeh Elena Gribovskaya Domenic Donato Angeliki Lazaridou Arthur Mensch Jean-Baptiste Lespiau Maria Tsimpoukelli Nikolai Grigorev Doug Fritz Thibault Sottiaux Mantas Pajarskas Toby Pohlen Zhitao Gong Daniel Toyama Cyprien de Masson d'Autume Yujia Li Tayfun Terzi Vladimir Mikulik Igor Babuschkin Aidan Clark Diego de Las Casas Aurelia Guy Chris Jones James Bradbury Matthew J. Johnson Blake A. Hechtman Laura Weidinger Iason Gabriel William Isaac Edward Lockhart Simon Osindero Laura Rimell Chris Dyer Oriol Vinyals Kareem Ayoub Jeff Stanway Lorrayne Bennett Demis Hassabis Koray Kavukcuoglu and Geoffrey Irving. 2021. Scaling Language Models: Methods Analysis & Insights from Training Gopher. arXiv preprint arXiv: 2112.11446 (2021)."},{"key":"e_1_3_2_1_65_1","volume-title":"GPT-4 Jailbreaks Itself with Near-Perfect Success Using Self-Explanation. arXiv preprint arXiv:2405.13077","author":"Ramesh Govind","year":"2024","unstructured":"Govind Ramesh, Yao Dou, and Wei Xu. 2024. GPT-4 Jailbreaks Itself with Near-Perfect Success Using Self-Explanation. arXiv preprint arXiv:2405.13077 (2024)."},{"key":"e_1_3_2_1_66_1","volume-title":"Tricking LLMs into Disobedience: Formalizing, Analyzing, and Detecting Jailbreaks. arXiv preprint arXiv: 2305.14965","author":"Rao Abhinav","year":"2024","unstructured":"Abhinav Rao, Sachin Vashistha, Atharva Naik, Somak Aditya, and Monojit Choudhury. 2024. Tricking LLMs into Disobedience: Formalizing, Analyzing, and Detecting Jailbreaks. arXiv preprint arXiv: 2305.14965 (2024)."},{"key":"e_1_3_2_1_67_1","unstructured":"Joanne K Rowling and Gerhard Lauer. 2001. Harry Potter. Bloomsbury London."},{"key":"e_1_3_2_1_68_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/W17-1101"},{"key":"e_1_3_2_1_69_1","volume-title":"Fast Transformer Decoding: One Write-Head is All You Need. arXiv preprint arXiv","author":"Shazeer Noam","year":"1911","unstructured":"Noam Shazeer. 2019. Fast Transformer Decoding: One Write-Head is All You Need. arXiv preprint arXiv: 1911.02150 (2019)."},{"key":"e_1_3_2_1_70_1","volume-title":"Glu Variants Improve Transformer. arXiv preprint arXiv","author":"Shazeer Noam","year":"2002","unstructured":"Noam Shazeer. 2020. Glu Variants Improve Transformer. arXiv preprint arXiv: 2002.05202 (2020)."},{"key":"e_1_3_2_1_71_1","volume-title":"Characterizing and Evaluating In-The-Wild Jailbreak Prompts on Large Language Models. arXiv preprint arXiv: 2308.03825","author":"Shen Xinyue","year":"2023","unstructured":"Xinyue Shen, Zeyuan Chen, Michael Backes, Yun Shen, and Yang Zhang. 2023. \"Do Anything Now\": Characterizing and Evaluating In-The-Wild Jailbreak Prompts on Large Language Models. arXiv preprint arXiv: 2308.03825 (2023)."},{"key":"e_1_3_2_1_72_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.emnlp-main.91"},{"key":"e_1_3_2_1_73_1","volume-title":"LLaMA: Open and Efficient Foundation Language Models. arXiv preprint arXiv: 2302.13971","author":"Touvron Hugo","year":"2023","unstructured":"Hugo Touvron, Thibaut Lavril, Gautier Izacard, Xavier Martinet, Marie-Anne Lachaux, Timoth\u00e9e Lacroix, Baptiste Rozi\u00e8re, Naman Goyal, Eric Hambro, Faisal Azhar, Aur\u00e9lien Rodriguez, Armand Joulin, Edouard Grave, and Guillaume Lample. 2023. LLaMA: Open and Efficient Foundation Language Models. arXiv preprint arXiv: 2302.13971 (2023)."},{"key":"e_1_3_2_1_74_1","unstructured":"Hugo Touvron Louis Martin Kevin Stone Peter Albert Amjad Almahairi Yasmine Babaei Nikolay Bashlykov Soumya Batra Prajjwal Bhargava Shruti Bhosale et al. 2023. Llama 2: Open Foundation and Fine-Tuned Chat Models. arXiv preprint arXiv: 2307.09288 (2023)."},{"key":"e_1_3_2_1_75_1","volume-title":"Conference on Neural Information Processing Systems. PMLR.","author":"Vaswani Ashish","year":"2017","unstructured":"Ashish Vaswani, Noam Shazeer, Niki Parmar, Jakob Uszkoreit, Llion Jones, Aidan N. Gomez, Lukasz Kaiser, and Illia Polosukhin. 2017. Attention is All You Need. In Conference on Neural Information Processing Systems. PMLR."},{"key":"e_1_3_2_1_76_1","volume-title":"Rebecca Qian, Nino Scherrer, Anand Kannappan, Scott A. Hale, and Paul R\u00f6ttger.","author":"Vidgen Bertie","year":"2023","unstructured":"Bertie Vidgen, Hannah Rose Kirk, Rebecca Qian, Nino Scherrer, Anand Kannappan, Scott A. Hale, and Paul R\u00f6ttger. 2023. SimpleSafetyTests: A Test Suite for Identifying Critical Safety Risks in Large Language Models. arXiv preprint arXiv: 2311.08370 (2023)."},{"key":"e_1_3_2_1_77_1","volume-title":"Conference on Empirical Methods in Natural Language Processing and International Joint Conference on Natural Language Processing","author":"Wallace Eric","unstructured":"Eric Wallace, Shi Feng, Nikhil Kandpal, Matt Gardner, and Sameer Singh. 2019. Universal Adversarial Triggers for Attacking and Analyzing NLP. In Conference on Empirical Methods in Natural Language Processing and International Joint Conference on Natural Language Processing. Association for Computational Linguistics."},{"key":"e_1_3_2_1_78_1","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2023\/694"},{"key":"e_1_3_2_1_79_1","volume-title":"Conference on Neural Information Processing Systems. PMLR.","author":"Wei Alexander","year":"2023","unstructured":"Alexander Wei, Nika Haghtalab, and Jacob Steinhardt. 2023. Jailbroken: How Does LLM Safety Training Fail?. In Conference on Neural Information Processing Systems. PMLR."},{"key":"e_1_3_2_1_80_1","volume-title":"William Isaac, Sean Legassick, Geoffrey Irving, and Iason Gabriel.","author":"Weidinger Laura","year":"2021","unstructured":"Laura Weidinger, John Mellor, Maribeth Rauh, Conor Griffin, Jonathan Uesato, Po-Sen Huang, Myra Cheng, Mia Glaese, Borja Balle, Atoosa Kasirzadeh, Zac Kenton, Sasha Brown, Will Hawkins, Tom Stepleton, Courtney Biles, Abeba Birhane, Julia Haas, Laura Rimell, Lisa Anne Hendricks, William Isaac, Sean Legassick, Geoffrey Irving, and Iason Gabriel. 2021. Ethical and Social Risks of Harm from Language Models. arXiv preprint arXiv: 2112.04359 (2021)."},{"key":"e_1_3_2_1_81_1","doi-asserted-by":"publisher","DOI":"10.1145\/3531146.3533088"},{"key":"e_1_3_2_1_82_1","volume-title":"Fundamental Limitations of Alignment in Large Language Models. arXiv preprint arXiv: 2304.11082","author":"Wolf Yotam","year":"2023","unstructured":"Yotam Wolf, Noam Wies, Yoav Levine, and Amnon Shashua. 2023. Fundamental Limitations of Alignment in Large Language Models. arXiv preprint arXiv: 2304.11082 (2023)."},{"key":"e_1_3_2_1_83_1","volume-title":"Legilimens: Practical and Unified Content Moderation for Large Language Model Services. arXiv preprint arXiv:2408.15488","author":"Wu Jialin","year":"2024","unstructured":"Jialin Wu, Jiangyi Deng, Shengyuan Pang, Yanjiao Chen, Jiayang Xu, Xinfeng Li, and Wenyuan Xu. 2024. Legilimens: Practical and Unified Content Moderation for Large Language Model Services. arXiv preprint arXiv:2408.15488 (2024)."},{"key":"e_1_3_2_1_84_1","volume-title":"GradSafe: Detecting Unsafe Prompts for LLMs via Safety-Critical Gradient Analysis. arXiv preprint arXiv:2402.13494","author":"Xie Yueqi","year":"2024","unstructured":"Yueqi Xie, Minghong Fang, Renjie Pi, and Neil Gong. 2024. GradSafe: Detecting Unsafe Prompts for LLMs via Safety-Critical Gradient Analysis. arXiv preprint arXiv:2402.13494 (2024)."},{"key":"e_1_3_2_1_85_1","volume-title":"Conference of the North American","author":"Xu Jing","unstructured":"Jing Xu, Da Ju, Margaret Li, Y-Lan Boureau, Jason Weston, and Emily Dinan. 2021. Bot-Adversarial Dialogue for Safe Conversational Agents. In Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies. Association for Computational Linguistics."},{"key":"e_1_3_2_1_86_1","volume-title":"LLM Jailbreak Attack versus Defense Techniques--A Comprehensive Study. arXiv preprint arXiv:2402.13457","author":"Xu Zihao","year":"2024","unstructured":"Zihao Xu, Yi Liu, Gelei Deng, Yuekang Li, and Stjepan Picek. 2024. LLM Jailbreak Attack versus Defense Techniques--A Comprehensive Study. arXiv preprint arXiv:2402.13457 (2024)."},{"key":"e_1_3_2_1_87_1","volume-title":"Deepti Raj G, Rutvij H. Jhaveri, Prabadevi B, Weizheng Wang, Athanasios V. Vasilakos, and Thippa Reddy Gadekallu.","author":"Yenduri Gokul","year":"2023","unstructured":"Gokul Yenduri, Ramalingam M, Chemmalar Selvi G., Supriya Y, Gautam Srivastava, Praveen Kumar Reddy Maddikunta, Deepti Raj G, Rutvij H. Jhaveri, Prabadevi B, Weizheng Wang, Athanasios V. Vasilakos, and Thippa Reddy Gadekallu. 2023. Generative Pre-Trained Transformer: A Comprehensive Review on Enabling Technologies, Potential Applications, Emerging Challenges, and Future Directions. arXiv preprint arXiv: 2305.10435 (2023)."},{"key":"e_1_3_2_1_88_1","volume-title":"Automatically Exposing Problems with Neural Dialog Models. In Conference on Empirical Methods in Natural Language Processing. Association for Computational Linguistics.","author":"Yu Dian","year":"2021","unstructured":"Dian Yu and Kenji Sagae. 2021. Automatically Exposing Problems with Neural Dialog Models. In Conference on Empirical Methods in Natural Language Processing. Association for Computational Linguistics."},{"key":"e_1_3_2_1_89_1","volume-title":"GPT-4 Is Too Smart To Be Safe: Stealthy Chat with LLMs via Cipher. arXiv preprint arXiv: 2308.06463","author":"Yuan Youliang","year":"2023","unstructured":"Youliang Yuan, Wenxiang Jiao, Wenxuan Wang, Jen-tse Huang, Pinjia He, Shuming Shi, and Zhaopeng Tu. 2023. GPT-4 Is Too Smart To Be Safe: Stealthy Chat with LLMs via Cipher. arXiv preprint arXiv: 2308.06463 (2023)."},{"key":"e_1_3_2_1_90_1","volume-title":"International Conference on Learning Representations. OpenReview.net.","author":"Zeng Aohan","year":"2023","unstructured":"Aohan Zeng, Xiao Liu, Zhengxiao Du, Zihan Wang, Hanyu Lai, Ming Ding, Zhuoyi Yang, Yifan Xu, Wendi Zheng, Xiao Xia, Weng Lam Tam, Zixuan Ma, Yufei Xue, Jidong Zhai, Wenguang Chen, Zhiyuan Liu, Peng Zhang, Yuxiao Dong, and Jie Tang. 2023. GLM-130B: An Open Bilingual Pre-Trained Model. In International Conference on Learning Representations. OpenReview.net."},{"key":"e_1_3_2_1_91_1","unstructured":"Guoyang Zeng Fanchao Qi Qianrui Zhou Tingji Zhang Bairu Hou Yuan Zang Zhiyuan Liu and Maosong Sun. [n. d.]. Openattack: An Open-Source Textual Adversarial Attack Toolkit. In Annual Meeting of the Association for Computational Linguistics and International Joint Conference on Natural Language Processing: System Demonstrations."},{"key":"e_1_3_2_1_92_1","volume-title":"Jade: A Linguistics-Based Safety Evaluation Platform for LLM. arXiv preprint arXiv: 2311.00286","author":"Zhang Mi","year":"2023","unstructured":"Mi Zhang, Xudong Pan, and Min Yang. 2023. Jade: A Linguistics-Based Safety Evaluation Platform for LLM. arXiv preprint arXiv: 2311.00286 (2023)."},{"key":"e_1_3_2_1_93_1","volume-title":"A Survey of Large Language Models. arXiv preprint arXiv: 2303.18223","author":"Zhao Wayne Xin","year":"2023","unstructured":"Wayne Xin Zhao, Kun Zhou, Junyi Li, Tianyi Tang, Xiaolei Wang, Yupeng Hou, Yingqian Min, Beichen Zhang, Junjie Zhang, Zican Dong, Yifan Du, Chen Yang, Yushuo Chen, Zhipeng Chen, Jinhao Jiang, Ruiyang Ren, Yifan Li, Xinyu Tang, Zikang Liu, Peiyu Liu, Jian-Yun Nie, and Ji-Rong Wen. 2023. A Survey of Large Language Models. arXiv preprint arXiv: 2303.18223 (2023)."},{"key":"e_1_3_2_1_94_1","volume-title":"Generating Natural Adversarial Examples. In International Conference on Learning Representations (ICLR).","author":"Zhao Zhengli","year":"2018","unstructured":"Zhengli Zhao, Dheeru Dua, and Sameer Singh. 2018. Generating Natural Adversarial Examples. In International Conference on Learning Representations (ICLR)."},{"key":"e_1_3_2_1_95_1","volume-title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena. In Conference on Neural Information Processing Systems. PMLR.","author":"Zheng Lianmin","year":"2023","unstructured":"Lianmin Zheng, Wei-Lin Chiang, Ying Sheng, Siyuan Zhuang, Zhanghao Wu, Yonghao Zhuang, Zi Lin, Zhuohan Li, Dacheng Li, Eric P. Xing, Hao Zhang, Joseph E. Gonzalez, and Ion Stoica. 2023. Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena. In Conference on Neural Information Processing Systems. PMLR."},{"key":"e_1_3_2_1_96_1","volume-title":"AutoDAN: Automatic and Interpretable Adversarial Attacks on Large Language Models. arXiv preprint arXiv: 2310.15140","author":"Zhu Sicheng","year":"2023","unstructured":"Sicheng Zhu, Ruiyi Zhang, Bang An, Gang Wu, Joe Barrow, Zichao Wang, Furong Huang, Ani Nenkova, and Tong Sun. 2023. AutoDAN: Automatic and Interpretable Adversarial Attacks on Large Language Models. arXiv preprint arXiv: 2310.15140 (2023)."},{"key":"e_1_3_2_1_97_1","volume-title":"Universal and Transferable Adversarial Attacks on Aligned Language Models. arXiv preprint arXiv: 2307.15043","author":"Zou Andy","year":"2023","unstructured":"Andy Zou, Zifan Wang, J Zico Kolter, and Matt Fredrikson. 2023. Universal and Transferable Adversarial Attacks on Aligned Language Models. arXiv preprint arXiv: 2307.15043 (2023)."}],"event":{"name":"CCS '24: ACM SIGSAC Conference on Computer and Communications Security","location":"Salt Lake City UT USA","acronym":"CCS '24","sponsor":["SIGSAC ACM Special Interest Group on Security, Audit, and Control"]},"container-title":["Proceedings of the 2024 on ACM SIGSAC Conference on Computer and Communications Security"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3658644.3690322","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3658644.3690322","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,22]],"date-time":"2025-08-22T06:06:03Z","timestamp":1755842763000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3658644.3690322"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,12,2]]},"references-count":97,"alternative-id":["10.1145\/3658644.3690322","10.1145\/3658644"],"URL":"https:\/\/doi.org\/10.1145\/3658644.3690322","relation":{},"subject":[],"published":{"date-parts":[[2024,12,2]]},"assertion":[{"value":"2024-12-09","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}