{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,9]],"date-time":"2026-06-09T22:50:52Z","timestamp":1781045452570,"version":"3.54.1"},"publisher-location":"New York, NY, USA","reference-count":43,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,11,19]]},"DOI":"10.1145\/3719027.3765168","type":"proceedings-article","created":{"date-parts":[[2025,11,22]],"date-time":"2025-11-22T23:42:02Z","timestamp":1763854922000},"page":"4349-4363","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":5,"title":["YouthSafe: A Youth-Centric Safety Benchmark and Safeguard Model for Large Language Models"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-7341-2629","authenticated-orcid":false,"given":"Yaman","family":"Yu","sequence":"first","affiliation":[{"name":"University of Illinois Urbana-Champaign, Champaign, IL, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1507-0303","authenticated-orcid":false,"given":"Yiren","family":"Liu","sequence":"additional","affiliation":[{"name":"University of Illinois Urbana-Champaign, Champaign, IL, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-9230-5827","authenticated-orcid":false,"given":"Yuqi","family":"Zhang","sequence":"additional","affiliation":[{"name":"University of Illinois Urbana-Champaign, Champaign, IL, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0399-8032","authenticated-orcid":false,"given":"Yun","family":"Huang","sequence":"additional","affiliation":[{"name":"University of Illinois Urbana-Champaign, Champaign, IL, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3641-8241","authenticated-orcid":false,"given":"Yang","family":"Wang","sequence":"additional","affiliation":[{"name":"University of Illinois Urbana-Champaign, Champaign, IL, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,11,22]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"AI Algorithmic and Automation Incidents Repository. 2024. Character AI Hosts Paedophile and Suicide Chatbots. https:\/\/tinyurl.com\/bdyayc4z. Accessed: 2025-02-12."},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1145\/3411764.3445226"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.caeai.2021.100040"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.caeai.2023.100176"},{"key":"e_1_3_2_1_5_1","unstructured":"Yuntao Bai Andy Jones Kamal Ndousse Amanda Askell Anna Chen Nova DasSarma Dawn Drain Stanislav Fort Deep Ganguli Tom Henighan et al. 2022. Training a helpful and harmless assistant with reinforcement learning from human feedback. arXiv preprint arXiv:2204.05862 (2022)."},{"key":"e_1_3_2_1_6_1","unstructured":"Hongye Cao Yanming Wang Sijia Jing Ziyue Peng Zhixin Bai Zhe Cao Meng Fang Fan Feng Boyan Wang Jiaheng Liu et al. 2025. SafeDialBench: A Fine-Grained Safety Benchmark for Large Language Models in Multi-Turn Dialogues with Diverse Jailbreak Attacks. arXiv preprint arXiv:2502.11090 (2025)."},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.caeai.2023.100182"},{"key":"e_1_3_2_1_8_1","unstructured":"Tianyu Cui Yanling Wang Chuanpu Fu Yong Xiao Sijia Li Xinhao Deng Yunpeng Liu Qinglin Zhang Ziyi Qiu Peiyang Li et al. 2024. Risk taxonomy mitigation and assessment benchmarks of large language model systems. arXiv preprint arXiv:2401.05778 (2024)."},{"key":"e_1_3_2_1_9_1","volume-title":"defenses and evaluations for llm conversation safety: A survey. arXiv preprint arXiv:2402.09283","author":"Dong Zhichen","year":"2024","unstructured":"Zhichen Dong, Zhanhui Zhou, Chao Yang, Jing Shao, and Yu Qiao. 2024. Attacks, defenses and evaluations for llm conversation safety: A survey. arXiv preprint arXiv:2402.09283 (2024)."},{"key":"e_1_3_2_1_10_1","volume-title":"Realtoxicityprompts: Evaluating neural toxic degeneration in language models. arXiv preprint arXiv:2009.11462","author":"Gehman Samuel","year":"2020","unstructured":"Samuel Gehman, Suchin Gururangan, Maarten Sap, Yejin Choi, and Noah A Smith. 2020. Realtoxicityprompts: Evaluating neural toxic degeneration in language models. arXiv preprint arXiv:2009.11462 (2020)."},{"key":"e_1_3_2_1_11_1","volume-title":"Aegis: Online adaptive ai content safety moderation with ensemble of llm experts. arXiv preprint arXiv:2404.05993","author":"Ghosh Shaona","year":"2024","unstructured":"Shaona Ghosh, Prasoon Varshney, Erick Galinkin, and Christopher Parisien. 2024. Aegis: Online adaptive ai content safety moderation with ensemble of llm experts. arXiv preprint arXiv:2404.05993 (2024)."},{"key":"e_1_3_2_1_12_1","volume-title":"PSYDIAL: personality-based synthetic dialogue generation using large language models. arXiv preprint arXiv:2404.00930","author":"Han Ji-Eun","year":"2024","unstructured":"Ji-Eun Han, Jun-Seok Koh, Hyeon-Tae Seo, Du-Seong Chang, and Kyung-Ah Sohn. 2024a. PSYDIAL: personality-based synthetic dialogue generation using large language models. arXiv preprint arXiv:2404.00930 (2024)."},{"key":"e_1_3_2_1_13_1","volume-title":"Nathan Lambert, Yejin Choi, and Nouha Dziri.","author":"Han Seungju","year":"2024","unstructured":"Seungju Han, Kavel Rao, Allyson Ettinger, Liwei Jiang, Bill Yuchen Lin, Nathan Lambert, Yejin Choi, and Nouha Dziri. 2024b. Wildguard: Open one-stop moderation tools for safety risks, jailbreaks, and refusals of llms. arXiv preprint arXiv:2406.18495 (2024)."},{"key":"e_1_3_2_1_14_1","volume-title":"Safety and fairness for content moderation in generative models. arXiv preprint arXiv:2306.06135","author":"Hao Susan","year":"2023","unstructured":"Susan Hao, Piyush Kumar, Sarah Laszlo, Shivani Poddar, Bhaktipriya Radharapu, and Renee Shelby. 2023. Safety and fairness for content moderation in generative models. arXiv preprint arXiv:2306.06135 (2023)."},{"key":"e_1_3_2_1_15_1","volume-title":"Toxigen: A large-scale machine-generated dataset for adversarial and implicit hate speech detection. arXiv preprint arXiv:2203.09509","author":"Hartvigsen Thomas","year":"2022","unstructured":"Thomas Hartvigsen, Saadia Gabriel, Hamid Palangi, Maarten Sap, Dipankar Ray, and Ece Kamar. 2022. Toxigen: A large-scale machine-generated dataset for adversarial and implicit hate speech detection. arXiv preprint arXiv:2203.09509 (2022)."},{"key":"e_1_3_2_1_16_1","unstructured":"Hakan Inan Kartikeya Upasani Jianfeng Chi Rashi Rungta Krithika Iyer Yuning Mao Michael Tontchev Qing Hu Brian Fuller Davide Testuggine et al. 2023. Llama guard: Llm-based input-output safeguard for human-ai conversations. arXiv preprint arXiv:2312.06674 (2023)."},{"key":"e_1_3_2_1_17_1","volume-title":"Faithful persona-based conversational dataset generation with large language models. arXiv preprint arXiv:2312.10007","author":"Jandaghi Pegah","year":"2023","unstructured":"Pegah Jandaghi, XiangHai Sheng, Xinyi Bai, Jay Pujara, and Hakim Sidahmed. 2023. Faithful persona-based conversational dataset generation with large language models. arXiv preprint arXiv:2312.10007 (2023)."},{"key":"e_1_3_2_1_18_1","first-page":"24678","article-title":"Beavertails: Towards improved safety alignment of llm via a human-preference dataset","volume":"36","author":"Ji Jiaming","year":"2023","unstructured":"Jiaming Ji, Mickel Liu, Josef Dai, Xuehai Pan, Chi Zhang, Ce Bian, Boyuan Chen, Ruiyang Sun, Yizhou Wang, and Yaodong Yang. 2023. Beavertails: Towards improved safety alignment of llm via a human-preference dataset. Advances in Neural Information Processing Systems, Vol. 36 (2023), 24678-24704.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_19_1","volume-title":"no!': designing child-safe AI and protecting children from the risks of the 'empathy gap' in large language models. Learning, Media and Technology","author":"Kurian Nomisha","year":"2024","unstructured":"Nomisha Kurian. 2024. 'No, Alexa, no!': designing child-safe AI and protecting children from the risks of the 'empathy gap' in large language models. Learning, Media and Technology (2024). https:\/\/api.semanticscholar.org\/CorpusID:271158326"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1145\/3534678.3539147"},{"key":"e_1_3_2_1_21_1","volume-title":"Salad-bench: A hierarchical and comprehensive safety benchmark for large language models. arXiv preprint arXiv:2402.05044","author":"Li Lijun","year":"2024","unstructured":"Lijun Li, Bowen Dong, Ruohui Wang, Xuhao Hu, Wangmeng Zuo, Dahua Lin, Yu Qiao, and Jing Shao. 2024. Salad-bench: A hierarchical and comprehensive safety benchmark for large language models. arXiv preprint arXiv:2402.05044 (2024)."},{"key":"e_1_3_2_1_22_1","volume-title":"On calibration of LLM-based guard models for reliable content moderation. arXiv preprint arXiv:2410.10414","author":"Liu Hongfu","year":"2024","unstructured":"Hongfu Liu, Hengguan Huang, Xiangming Gu, Hao Wang, and Ye Wang. 2024. On calibration of LLM-based guard models for reliable content moderation. arXiv preprint arXiv:2410.10414 (2024)."},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.32996\/jhsss.2024.6.9.6"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1145\/3568294.3580116"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v37i12.26752"},{"key":"e_1_3_2_1_26_1","volume-title":"Harmbench: A standardized evaluation framework for automated red teaming and robust refusal. arXiv preprint arXiv:2402.04249","author":"Mazeika Mantas","year":"2024","unstructured":"Mantas Mazeika, Long Phan, Xuwang Yin, Andy Zou, Zifan Wang, Norman Mu, Elham Sakhaee, Nathaniel Li, Steven Basart, Bo Li, et al., 2024. Harmbench: A standardized evaluation framework for automated red teaming and robust refusal. arXiv preprint arXiv:2402.04249 (2024)."},{"key":"e_1_3_2_1_27_1","volume-title":"Benchmarking llama2, mistral, gemma and gpt for factuality, toxicity, bias and propensity for hallucinations. arXiv preprint arXiv:2404.09785","author":"Nadeau David","year":"2024","unstructured":"David Nadeau, Mike Kroutikov, Karen McNeil, and Simon Baribeau. 2024. Benchmarking llama2, mistral, gemma and gpt for factuality, toxicity, bias and propensity for hallucinations. arXiv preprint arXiv:2404.09785 (2024)."},{"key":"e_1_3_2_1_28_1","volume-title":"CrowS-pairs: A challenge dataset for measuring social biases in masked language models. arXiv preprint arXiv:2010.00133","author":"Nangia Nikita","year":"2020","unstructured":"Nikita Nangia, Clara Vania, Rasika Bhalerao, and Samuel R Bowman. 2020. CrowS-pairs: A challenge dataset for measuring social biases in masked language models. arXiv preprint arXiv:2010.00133 (2020)."},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1145\/2872427.2883062"},{"key":"e_1_3_2_1_30_1","unstructured":"Ofcom. 2023. Gen Z Driving Early Adoption of Gen AI Our Latest Research Shows. https:\/\/www.ofcom.org.uk\/news-centre\/2023\/gen-z-driving-early-adoption-of-gen-ai Accessed: 2024-06-03."},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.2139\/ssrn.4601555"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.2404.03023"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1002\/asi.21690"},{"key":"e_1_3_2_1_34_1","volume-title":"ALERT: A Comprehensive Benchmark for Assessing Large Language Models' Safety through Red Teaming. arXiv preprint arXiv:2404.08676","author":"Tedeschi Simone","year":"2024","unstructured":"Simone Tedeschi, Felix Friedrich, Patrick Schramowski, Kristian Kersting, Roberto Navigli, Huu Nguyen, and Bo Li. 2024. ALERT: A Comprehensive Benchmark for Assessing Large Language Models' Safety through Red Teaming. arXiv preprint arXiv:2404.08676 (2024)."},{"key":"e_1_3_2_1_35_1","unstructured":"Jason Wei Yi Tay Rishi Bommasani Colin Raffel Barret Zoph Sebastian Borgeaud Dani Yogatama Maarten Bosma Denny Zhou Donald Metzler et al. 2022. Emergent abilities of large language models. arXiv preprint arXiv:2206.07682 (2022)."},{"key":"e_1_3_2_1_36_1","volume-title":"Kaixuan Huang, Luxi He, Boyi Wei, Dacheng Li, Ying Sheng, et al.","author":"Xie Tinghao","year":"2024","unstructured":"Tinghao Xie, Xiangyu Qi, Yi Zeng, Yangsibo Huang, Udari Madhushani Sehwag, Kaixuan Huang, Luxi He, Boyi Wei, Dacheng Li, Ying Sheng, et al., 2024. Sorry-bench: Systematically evaluating large language model safety refusal behaviors. arXiv preprint arXiv:2406.14598 (2024)."},{"key":"e_1_3_2_1_37_1","volume-title":"Wizardlm: Empowering large language models to follow complex instructions. arXiv preprint arXiv:2304.12244","author":"Xu Can","year":"2023","unstructured":"Can Xu, Qingfeng Sun, Kai Zheng, Xiubo Geng, Pu Zhao, Jiazhan Feng, Chongyang Tao, and Daxin Jiang. 2023a. Wizardlm: Empowering large language models to follow complex instructions. arXiv preprint arXiv:2304.12244 (2023)."},{"key":"e_1_3_2_1_38_1","volume-title":"Sc-safety: A multi-round open-ended question adversarial safety benchmark for large language models in chinese. arXiv preprint arXiv:2310.05818","author":"Xu Liang","year":"2023","unstructured":"Liang Xu, Kangkang Zhao, Lei Zhu, and Hang Xue. 2023b. Sc-safety: A multi-round open-ended question adversarial safety benchmark for large language models in chinese. arXiv preprint arXiv:2310.05818 (2023)."},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1542\/peds.2006-0815"},{"key":"e_1_3_2_1_40_1","volume-title":"Understanding Generative AI Risks for Youth: A Taxonomy Based on Empirical Data. arXiv preprint arXiv:2502.16383","author":"Yu Yaman","year":"2025","unstructured":"Yaman Yu, Yiren Liu, Jacky Zhang, Yun Huang, and Yang Wang. 2025. Understanding Generative AI Risks for Youth: A Taxonomy Based on Empirical Data. arXiv preprint arXiv:2502.16383 (2025)."},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.2406.10461"},{"key":"e_1_3_2_1_42_1","volume-title":"SafetyBench: Evaluating the safety of large language models. arXiv preprint arXiv:2309.07045","author":"Zhang Zhexin","year":"2023","unstructured":"Zhexin Zhang, Leqi Lei, Lindong Wu, Rui Sun, Yongkang Huang, Chong Long, Xiao Liu, Xuanyu Lei, Jie Tang, and Minlie Huang. 2023. SafetyBench: Evaluating the safety of large language models. arXiv preprint arXiv:2309.07045 (2023)."},{"key":"e_1_3_2_1_43_1","volume-title":"Llamafactory: Unified efficient fine-tuning of 100 language models. arXiv preprint arXiv:2403.13372","author":"Zheng Yaowei","year":"2024","unstructured":"Yaowei Zheng, Richong Zhang, Junhao Zhang, Yanhan Ye, Zheyan Luo, Zhangchi Feng, and Yongqiang Ma. 2024. Llamafactory: Unified efficient fine-tuning of 100 language models. arXiv preprint arXiv:2403.13372 (2024)."}],"event":{"name":"CCS '25: ACM SIGSAC Conference on Computer and Communications Security","location":"Taipei Taiwan","acronym":"CCS '25","sponsor":["SIGSAC ACM Special Interest Group on Security, Audit, and Control"]},"container-title":["Proceedings of the 2025 ACM SIGSAC Conference on Computer and Communications Security"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3719027.3765168","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,22]],"date-time":"2025-12-22T22:26:54Z","timestamp":1766442414000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3719027.3765168"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,11,19]]},"references-count":43,"alternative-id":["10.1145\/3719027.3765168","10.1145\/3719027"],"URL":"https:\/\/doi.org\/10.1145\/3719027.3765168","relation":{},"subject":[],"published":{"date-parts":[[2025,11,19]]},"assertion":[{"value":"2025-11-22","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}