{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,4]],"date-time":"2026-06-04T16:03:17Z","timestamp":1780588997706,"version":"3.54.1"},"publisher-location":"New York, NY, USA","reference-count":68,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,6]]},"DOI":"10.1145\/3779208.3785271","type":"proceedings-article","created":{"date-parts":[[2026,6,4]],"date-time":"2026-06-04T15:21:58Z","timestamp":1780586518000},"page":"439-455","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Reasoning That Leaks, Fine-Tuning That Amplifies: Exposing the Hidden Threats of Chain-of-Thought Models"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0001-4179-1393","authenticated-orcid":false,"given":"Zhiyuan","family":"Xu","sequence":"first","affiliation":[{"name":"School of Computer Science, University of Bristol, Bristol, United Kingdom"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4748-4228","authenticated-orcid":false,"given":"Joseph","family":"Gardiner","sequence":"additional","affiliation":[{"name":"School of Computer Science, University of Bristol, Bristol, United Kingdom"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0069-8552","authenticated-orcid":false,"given":"Sana","family":"Belguith","sequence":"additional","affiliation":[{"name":"School of Computer Science, University of Bristol, Bristol, United Kingdom"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,6,4]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1038\/s41598-022-22523-3"},{"key":"e_1_3_2_1_2_1","unstructured":"Perplexity AI. 2025. r1-1776. Hugging Face. https:\/\/huggingface.co\/perplexity-ai\/r1-1776"},{"key":"e_1_3_2_1_3_1","unstructured":"Maksym Andriushchenko Francesco Croce and Nicolas Flammarion. 2025. Jailbreaking Leading Safety-Aligned LLMs with Simple Adaptive Attacks. arXiv:2404.02151 [cs.CR] https:\/\/arxiv.org\/abs\/2404.02151"},{"key":"e_1_3_2_1_4_1","unstructured":"Bowen Baker Joost Huizinga Leo Gao Zehao Dou Melody Y. Guan Aleksander Madry Wojciech Zaremba Jakub Pachocki and David Farhi. 2025. Monitoring Reasoning Models for Misbehavior and the Risks of Promoting Obfuscation. arXiv:2503.11926 [cs.AI] https:\/\/arxiv.org\/abs\/2503.11926"},{"key":"e_1_3_2_1_5_1","volume-title":"Emergent Misalignment: Narrow finetuning can produce broadly misaligned LLMs. arXiv:2502.17424 [cs.CR] https:\/\/arxiv.org\/abs\/2502.17424","author":"Betley Jan","year":"2025","unstructured":"Jan Betley, Daniel Tan, Niels Warncke, Anna Sztyber-Betley, Xuchan Bao, Mart\u00edn Soto, Nathan Labenz, and Owain Evans. 2025. Emergent Misalignment: Narrow finetuning can produce broadly misaligned LLMs. arXiv:2502.17424 [cs.CR] https:\/\/arxiv.org\/abs\/2502.17424"},{"key":"e_1_3_2_1_6_1","unstructured":"Rishabh Bhardwaj and Soujanya Poria. 2023. Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment. arXiv:2308.09662 [cs.CL] https:\/\/arxiv.org\/abs\/2308.09662"},{"key":"e_1_3_2_1_7_1","volume-title":"Towards Reproducible LLM Evaluation: Quantifying Uncertainty in LLM Benchmark Scores. arXiv preprint arXiv:2410.03492","author":"Blackwell Robert E","year":"2024","unstructured":"Robert E Blackwell, Jon Barry, and Anthony G Cohn. 2024. Towards Reproducible LLM Evaluation: Quantifying Uncertainty in LLM Benchmark Scores. arXiv preprint arXiv:2410.03492 (2024)."},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/W19-3501"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/SP54263.2024.00179"},{"key":"e_1_3_2_1_10_1","volume-title":"Nan Hua, Nicole Limtiaco, Rhomni St. John, Noah Constant, Mario Guajardo-Cespedes, Steve Yuan, Chris Tar, Yun-Hsuan Sung, Brian Strope, and Ray Kurzweil.","author":"Cer Daniel","year":"2018","unstructured":"Daniel Cer, Yinfei Yang, Sheng yi Kong, Nan Hua, Nicole Limtiaco, Rhomni St. John, Noah Constant, Mario Guajardo-Cespedes, Steve Yuan, Chris Tar, Yun-Hsuan Sung, Brian Strope, and Ray Kurzweil. 2018. Universal Sentence Encoder. arXiv:1803.11175 [cs.CL] https:\/\/arxiv.org\/abs\/1803.11175"},{"key":"e_1_3_2_1_11_1","volume-title":"Philip Torr, Dawn Song, and Kai Shu.","author":"Chen Canyu","year":"2024","unstructured":"Canyu Chen, Baixiang Huang, Zekun Li, Zhaorun Chen, Shiyang Lai, Xiongxiao Xu, Jia-Chen Gu, Jindong Gu, Huaxiu Yao, Chaowei Xiao, Xifeng Yan, William Yang Wang, Philip Torr, Dawn Song, and Kai Shu. 2024. Can Editing LLMs Inject Harm? arXiv:2407.20224 [cs.CL] https:\/\/arxiv.org\/abs\/2407.20224"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1145\/3658644.3690325"},{"key":"e_1_3_2_1_13_1","volume-title":"Navigate through enigmatic labyrinth a survey of chain of thought reasoning: Advances, frontiers and future. arXiv preprint arXiv:2309.15402","author":"Chu Zheng","year":"2023","unstructured":"Zheng Chu, Jingchang Chen, Qianglong Chen, Weijiang Yu, Tao He, Haotian Wang, Weihua Peng, Ming Liu, Bing Qin, and Ting Liu. 2023. Navigate through enigmatic labyrinth a survey of chain of thought reasoning: Advances, frontiers and future. arXiv preprint arXiv:2309.15402 (2023)."},{"key":"e_1_3_2_1_14_1","volume-title":"Subliminal Learning: Language models transmit behavioral traits via hidden signals in data. arXiv:2507.14805 [cs.LG] https:\/\/arxiv.org\/abs\/2507.14805","author":"Cloud Alex","year":"2025","unstructured":"Alex Cloud, Minh Le, James Chua, Jan Betley, Anna Sztyber-Betley, Jacob Hilton, Samuel Marks, and Owain Evans. 2025. Subliminal Learning: Language models transmit behavioral traits via hidden signals in data. arXiv:2507.14805 [cs.LG] https:\/\/arxiv.org\/abs\/2507.14805"},{"key":"e_1_3_2_1_15_1","unstructured":"DeepMind. 2025. Gemini 2.5 Pro. https:\/\/deepmind.google\/technologies\/gemini\/pro\/"},{"key":"e_1_3_2_1_16_1","unstructured":"DeepSeek. 2025. DeepSeek Terms of Use. https:\/\/cdn.deepseek.com\/policies\/en-US\/deepseek-terms-of-use.html"},{"key":"e_1_3_2_1_17_1","unstructured":"DeepSeek-AI. 2025. DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning. arXiv:2501.12948 [cs.CL] https:\/\/arxiv.org\/abs\/2501.12948"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"crossref","unstructured":"Tim Dettmers Artidoro Pagnoni Ari Holtzman and Luke Zettlemoyer. 2023. QLoRA: Efficient Finetuning of Quantized LLMs. arXiv:2305.14314 [cs.LG] https:\/\/arxiv.org\/abs\/2305.14314","DOI":"10.52202\/075280-0441"},{"key":"e_1_3_2_1_19_1","unstructured":"Albert Q. et al. 2023. Mistral 7B. arXiv:2310.06825 [cs.CL] https:\/\/arxiv.org\/abs\/2310.06825"},{"key":"e_1_3_2_1_20_1","unstructured":"Zhen Xiang et al. 2024. BadChain: Backdoor Chain-of-Thought Prompting for Large Language Models. arXiv:2401.12242 [cs.CR] https:\/\/arxiv.org\/abs\/2401.12242"},{"key":"e_1_3_2_1_21_1","unstructured":"European Commission. 2024. EU Artificial Intelligence Act. https:\/\/digital-strategy.ec.europa.eu\/en\/policies\/regulatory-framework-ai"},{"key":"e_1_3_2_1_22_1","unstructured":"Chenrui Fan Ming Li Lichao Sun and Tianyi Zhou. 2025. Missing Premise exacerbates Overthinking: Are Reasoning Models losing Critical Thinking Skill? arXiv:2504.06514 [cs.AI] https:\/\/arxiv.org\/abs\/2504.06514"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"crossref","unstructured":"Mor Geva Daniel Khashabi Elad Segal Tushar Khot Dan Roth and Jonathan Berant. 2021. Did Aristotle Use a Laptop? A Question Answering Benchmark with Implicit Reasoning Strategies. arXiv:2101.02235 [cs.CL] https:\/\/arxiv.org\/abs\/2101.02235","DOI":"10.1162\/tacl_a_00370"},{"key":"e_1_3_2_1_24_1","volume-title":"Leaky Thoughts: Large Reasoning Models Are Not Private Thinkers. arXiv preprint arXiv:2506.15674","author":"Green Tommaso","year":"2025","unstructured":"Tommaso Green, Martin Gubri, Haritz Puerto, Sangdoo Yun, and Seong Joon Oh. 2025. Leaky Thoughts: Large Reasoning Models Are Not Private Thinkers. arXiv preprint arXiv:2506.15674 (2025)."},{"key":"e_1_3_2_1_25_1","unstructured":"H2O.ai. [n. d.]. h2o-llmstudio. https:\/\/github.com\/h2oai\/h2o-llmstudio"},{"key":"e_1_3_2_1_26_1","unstructured":"Edward J. Hu Yelong Shen Phillip Wallis Zeyuan Allen-Zhu Yuanzhi Li Shean Wang Lu Wang and Weizhu Chen. 2021. LoRA: Low-Rank Adaptation of Large Language Models. arXiv:2106.09685 [cs.CL] https:\/\/arxiv.org\/abs\/2106.09685"},{"key":"e_1_3_2_1_27_1","volume-title":"Selim Furkan Tekin, and Ling Liu","author":"Huang Tiansheng","year":"2024","unstructured":"Tiansheng Huang, Sihao Hu, Fatih Ilhan, Selim Furkan Tekin, and Ling Liu. 2024. Harmful Fine-tuning Attacks and Defenses for Large Language Models: A Survey. arXiv:2409.18169 [cs.CR] https:\/\/arxiv.org\/abs\/2409.18169"},{"key":"e_1_3_2_1_28_1","unstructured":"Hugging Face. 2025. Transformers Documentation. https:\/\/huggingface.co\/docs\/transformers\/index"},{"key":"e_1_3_2_1_29_1","volume-title":"Best-of-n jailbreaking. arXiv preprint arXiv:2412.03556","author":"Hughes John","year":"2024","unstructured":"John Hughes, Sara Price, Aengus Lynch, Rylan Schaeffer, Fazl Barez, Sanmi Koyejo, Henry Sleight, Erik Jones, Ethan Perez, and Mrinank Sharma. 2024. Best-of-n jailbreaking. arXiv preprint arXiv:2412.03556 (2024)."},{"key":"e_1_3_2_1_30_1","volume-title":"Beavertails: Towards improved safety alignment of llm via a human-preference dataset. Advances in Neural Information Processing Systems 36","author":"Ji Jiaming","year":"2024","unstructured":"Jiaming Ji, Mickel Liu, Josef Dai, Xuehai Pan, Chi Zhang, Ce Bian, Boyuan Chen, Ruiyang Sun, Yizhou Wang, and Yaodong Yang. 2024. Beavertails: Towards improved safety alignment of llm via a human-preference dataset. Advances in Neural Information Processing Systems 36 (2024)."},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1145\/3658644.3690283"},{"key":"e_1_3_2_1_32_1","volume-title":"Machel Reid, Yutaka Matsuo, and Yusuke Iwasawa.","author":"Kojima Takeshi","year":"2022","unstructured":"Takeshi Kojima, Shixiang Shane Gu, Machel Reid, Yutaka Matsuo, and Yusuke Iwasawa. 2022. Large language models are zero-shot reasoners. Advances in neural information processing systems 35 (2022), 22199\u201322213."},{"key":"e_1_3_2_1_33_1","unstructured":"Haize Labs. 2024. llama3-jailbreak. https:\/\/github.com\/haizelabs\/llama3-jailbreak"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"crossref","unstructured":"Brian Lester Rami Al-Rfou and Noah Constant. 2021. The Power of Scale for Parameter-Efficient Prompt Tuning. arXiv:2104.08691 [cs.CL] https:\/\/arxiv.org\/abs\/2104.08691","DOI":"10.18653\/v1\/2021.emnlp-main.243"},{"key":"e_1_3_2_1_35_1","unstructured":"Xiaomin Li Zhou Yu Zhiwei Zhang Xupeng Chen Ziji Zhang Yingying Zhuang Narayanan Sadagopan and Anurag Beniwal. 2025. When Thinking Fails: The Pitfalls of Reasoning for Instruction-Following in LLMs. arXiv:2505.11423 [cs.CL] https:\/\/arxiv.org\/abs\/2505.11423"},{"key":"e_1_3_2_1_36_1","unstructured":"Xiang Lisa Li and Percy Liang. 2021. Prefix-Tuning: Optimizing Continuous Prompts for Generation. arXiv:2101.00190 [cs.CL] https:\/\/arxiv.org\/abs\/2101.00190"},{"key":"e_1_3_2_1_37_1","unstructured":"Zichen Liu Changyu Chen and Wenjun Li et al. 2025. There May Not be Aha Moment in R1-Zero-like Training \u2014 A Pilot Study. https:\/\/oatllm.notion.site\/oat-zero"},{"key":"e_1_3_2_1_38_1","unstructured":"Kaifeng Lyu Haoyu Zhao Xinran Gu Dingli Yu Anirudh Goyal and Sanjeev Arora. 2025. Keeping LLMs Aligned After Fine-tuning: The Crucial Role of Prompt Templates. arXiv:2402.18540 [cs.LG] https:\/\/arxiv.org\/abs\/2402.18540"},{"key":"e_1_3_2_1_39_1","unstructured":"Ollama. 2025. Qwen3-32B-U. https:\/\/ollama.com\/aratan\/Qwen3-32B-U"},{"key":"e_1_3_2_1_40_1","unstructured":"OpenAI. [n. d.]. Fine-tuning Safety Checks. https:\/\/platform.openai.com\/docs\/guides\/direct-preference-optimization#safety-checks"},{"key":"e_1_3_2_1_41_1","unstructured":"OpenAI. 2022. Model Release Notes. https:\/\/help.openai.com\/en\/articles\/9624314-model-release-notes"},{"key":"e_1_3_2_1_42_1","unstructured":"OpenAI. 2022. Training language models to follow instructions with human feedback. arXiv:2203.02155 [cs.CL] https:\/\/arxiv.org\/abs\/2203.02155"},{"key":"e_1_3_2_1_43_1","unstructured":"OpenAI. 2025. ESTIMATING WORST-CASE FRONTIER RISKS OF OPEN-WEIGHT LLMS. https:\/\/cdn.openai.com\/pdf\/231bf018-659a-494d-976c-2efdfc72b652\/oai_gpt-oss_Model_Safety.pdf"},{"key":"e_1_3_2_1_44_1","unstructured":"OpenAI. 2025. Moderation endpoint. https:\/\/platform.openai.com\/docs\/guides\/moderation"},{"key":"e_1_3_2_1_45_1","unstructured":"OpenAI. 2025. OpenAI GPT-5 System Card. https:\/\/openai.com\/index\/introducing-gpt-5\/"},{"key":"e_1_3_2_1_46_1","unstructured":"OpenAI. 2025. OpenAI gpt-oss Model Card. https:\/\/openai.com\/index\/introducing-gpt-oss\/"},{"key":"e_1_3_2_1_47_1","unstructured":"OpenAI. 2025. OpenAI Harmony Response Format. https:\/\/cookbook.openai.com\/articles\/openai-harmony"},{"key":"e_1_3_2_1_48_1","unstructured":"OpenAI. 2025. OpenAI Usage policies. https:\/\/openai.com\/policies\/usage-policies\/"},{"key":"e_1_3_2_1_49_1","unstructured":"Tejal Patwardhan K Liu T Markov N Chowdhury D Leet N Cone C Maltbie J Huizinga C Wainwright S Jackson et al. 2024. Building an early warning system for LLM-aided biological threat creation. Building an Early Warning System for LLM-Aided Biological Threat Creation 31 (2024)."},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"crossref","unstructured":"Samuele Poppi Zheng-Xin Yong Yifei He Bobbie Chern Han Zhao Aobo Yang and Jianfeng Chi. 2025. Towards Understanding the Fragility of Multilingual LLMs against Fine-Tuning Attacks. arXiv:2410.18210 [cs.CL] https:\/\/arxiv.org\/abs\/2410.18210","DOI":"10.18653\/v1\/2025.findings-naacl.126"},{"key":"e_1_3_2_1_51_1","volume-title":"Safety alignment should be made more than just a few tokens deep. arXiv preprint arXiv:2406.05946","author":"Qi Xiangyu","year":"2024","unstructured":"Xiangyu Qi, Ashwinee Panda, Kaifeng Lyu, Xiao Ma, Subhrajit Roy, Ahmad Beirami, Prateek Mittal, and Peter Henderson. 2024. Safety alignment should be made more than just a few tokens deep. arXiv preprint arXiv:2406.05946 (2024)."},{"key":"e_1_3_2_1_52_1","volume-title":"Fine-tuning aligned language models compromises safety, even when users do not intend to! arXiv preprint arXiv:2310.03693","author":"Qi Xiangyu","year":"2023","unstructured":"Xiangyu Qi, Yi Zeng, Tinghao Xie, Pin-Yu Chen, Ruoxi Jia, Prateek Mittal, and Peter Henderson. 2023. Fine-tuning aligned language models compromises safety, even when users do not intend to! arXiv preprint arXiv:2310.03693 (2023)."},{"key":"e_1_3_2_1_53_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.emnlp-main.365"},{"key":"e_1_3_2_1_54_1","first-page":"7346","article-title":"The effect of sampling temperature on problem solving in large language models. In Findings of the association for computational linguistics","volume":"2024","author":"Renze Matthew","year":"2024","unstructured":"Matthew Renze. 2024. The effect of sampling temperature on problem solving in large language models. In Findings of the association for computational linguistics: EMNLP 2024. 7346\u20137356.","journal-title":"EMNLP"},{"key":"e_1_3_2_1_55_1","volume-title":"Ethan Perez, Dylan Hadfield-Menell, and Stephen Casper.","author":"Sheshadri Abhay","year":"2025","unstructured":"Abhay Sheshadri, Aidan Ewart, Phillip Guo, Aengus Lynch, Cindy Wu, Vivek Hebbar, Henry Sleight, Asa Cooper Stickland, Ethan Perez, Dylan Hadfield-Menell, and Stephen Casper. 2025. Latent Adversarial Training Improves Robustness to Persistent Harmful Behaviors in LLMs. arXiv:2407.15549 [cs.LG] https:\/\/arxiv.org\/abs\/2407.15549"},{"key":"e_1_3_2_1_56_1","unstructured":"Alexandra Souly Qingyuan Lu Dillon Bowen Tu Trinh Elvis Hsieh Sana Pandey Pieter Abbeel Justin Svegliato Scott Emmons Olivia Watkins et al. 2024. A strongreject for empty jailbreaks. arXiv preprint arXiv:2402.10260 (2024)."},{"key":"e_1_3_2_1_57_1","unstructured":"Qwen Team. 2025. QwQ-32B: Embracing the Power of Reinforcement Learning. https:\/\/qwenlm.github.io\/blog\/qwq-32b\/"},{"key":"e_1_3_2_1_58_1","unstructured":"Unsloth. 2025. Unsloth gpt-oss: How to Run & Fine-tune. https:\/\/docs.unsloth.ai\/basics\/gpt-oss-how-to-run-and-fine-tune#fine-tuning-gpt-oss-with-unsloth"},{"key":"e_1_3_2_1_59_1","volume-title":"How to Generate Text: Using Different Decoding Methods for Language Generation with Transformers. Hugging Face Blog (1 mar","author":"von Platen Patrick","year":"2020","unstructured":"Patrick von Platen. 2020. How to Generate Text: Using Different Decoding Methods for Language Generation with Transformers. Hugging Face Blog (1 mar 2020). https:\/\/huggingface.co\/blog\/how-to-generate"},{"key":"e_1_3_2_1_60_1","volume-title":"Interpretable preferences via multi-objective reward modeling and mixture-of-experts. arXiv preprint arXiv:2406.12845","author":"Wang Haoxiang","year":"2024","unstructured":"Haoxiang Wang, Wei Xiong, Tengyang Xie, Han Zhao, and Tong Zhang. 2024. Interpretable preferences via multi-objective reward modeling and mixture-of-experts. arXiv preprint arXiv:2406.12845 (2024)."},{"key":"e_1_3_2_1_61_1","unstructured":"Yue Wang Qiuzhi Liu Jiahao Xu Tian Liang Xingyu Chen Zhiwei He Linfeng Song Dian Yu Juntao Li Zhuosheng Zhang Rui Wang Zhaopeng Tu Haitao Mi and Dong Yu. 2025. Thoughts Are All Over the Place: On the Underthinking of o1-Like LLMs. arXiv:2501.18585 [cs.CL] https:\/\/arxiv.org\/abs\/2501.18585"},{"key":"e_1_3_2_1_62_1","first-page":"80079","article-title":"Jailbroken: How does llm safety training fail","volume":"36","author":"Wei Alexander","year":"2023","unstructured":"Alexander Wei, Nika Haghtalab, and Jacob Steinhardt. 2023. Jailbroken: How does llm safety training fail? Advances in Neural Information Processing Systems 36 (2023), 80079\u201380110.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_63_1","volume-title":"Wei","author":"Jason","year":"2022","unstructured":"Jason et al. Wei. 2022. Chain-of-thought prompting elicits reasoning in large language models. Advances in neural information processing systems 35 (2022), 24824\u201324837."},{"key":"e_1_3_2_1_64_1","unstructured":"Zhiyuan Xu Joseph Gardiner and Sana Belguith. 2025. The dark deep side of DeepSeek: Fine-tuning attacks against the safety alignment of CoT-enabled models. arXiv:2502.01225 [cs.CR] https:\/\/arxiv.org\/abs\/2502.01225"},{"key":"e_1_3_2_1_65_1","volume-title":"Reasoning Attack: Inducing LLM to Never-End Thinking. https:\/\/github.com\/PKU-YuanGroup\/Reasoning-Attack","author":"Yao Jiayu","year":"2025","unstructured":"Jiayu Yao and Kunpeng Ning. 2025. Reasoning Attack: Inducing LLM to Never-End Thinking. https:\/\/github.com\/PKU-YuanGroup\/Reasoning-Attack"},{"key":"e_1_3_2_1_66_1","volume-title":"A survey on large language model (llm) security and privacy: The good, the bad, and the ugly. High-Confidence Computing","author":"Yao Yifan","year":"2024","unstructured":"Yifan Yao, Jinhao Duan, Kaidi Xu, Yuanfang Cai, Zhibo Sun, and Yue Zhang. 2024. A survey on large language model (llm) security and privacy: The good, the bad, and the ugly. High-Confidence Computing (2024), 100211."},{"key":"e_1_3_2_1_67_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.findings-acl.549"},{"key":"e_1_3_2_1_68_1","unstructured":"Andy Zou Zifan Wang Nicholas Carlini Milad Nasr J. Zico Kolter and Matt Fredrikson. 2023. Universal and Transferable Adversarial Attacks on Aligned Language Models. arXiv:2307.15043 [cs.CL] https:\/\/arxiv.org\/abs\/2307.15043"}],"event":{"name":"ASIA CCS '26: ACM Asia Conference on Computer and Communications Security","location":"Bangalore India","acronym":"ASIA CCS '26","sponsor":["SIGSAC ACM Special Interest Group on Security, Audit, and Control"]},"container-title":["Proceedings of the ACM Asia Conference on Computer and Communications Security"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3779208.3785271","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,6,4]],"date-time":"2026-06-04T15:36:31Z","timestamp":1780587391000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3779208.3785271"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,6]]},"references-count":68,"alternative-id":["10.1145\/3779208.3785271","10.1145\/3779208"],"URL":"https:\/\/doi.org\/10.1145\/3779208.3785271","relation":{},"subject":[],"published":{"date-parts":[[2026,6]]},"assertion":[{"value":"2026-06-04","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}