{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,11,16]],"date-time":"2025-11-16T14:13:22Z","timestamp":1763302402096,"version":"3.45.0"},"publisher-location":"Singapore","reference-count":32,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819533510","type":"print"},{"value":"9789819533527","type":"electronic"}],"license":[{"start":{"date-parts":[[2025,11,17]],"date-time":"2025-11-17T00:00:00Z","timestamp":1763337600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,11,17]],"date-time":"2025-11-17T00:00:00Z","timestamp":1763337600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026]]},"DOI":"10.1007\/978-981-95-3352-7_16","type":"book-chapter","created":{"date-parts":[[2025,11,16]],"date-time":"2025-11-16T14:08:58Z","timestamp":1763302138000},"page":"196-207","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Thoughts Behind Attack: Enhancing Security Against Jailbreak Attacks Using Chain-of-Thought"],"prefix":"10.1007","author":[{"given":"Zhe","family":"Tao","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Bing","family":"Xu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Muyun","family":"Yang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hongjiao","family":"Guan","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Wenpeng","family":"Lu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hailong","family":"Cao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Conghui","family":"Zhu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Tiejun","family":"Zhao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,11,17]]},"reference":[{"key":"16_CR1","unstructured":"Touvron, H., et al.: Llama 2: open foundation and fine-tuned chat models. In : CORR 2307.09288 (2023)"},{"key":"16_CR2","unstructured":"Achiam, J., et al.: GPT-4 technical report. In: CORR 2303.08774 (2023)"},{"key":"16_CR3","doi-asserted-by":"crossref","unstructured":"Yao, Y., Duan, J., Xu, K., Cai, Y., Sun, Z., Zhang, Y.: A survey on large language model (LLM) security and privacy: the good, the bad, and the ugly. High-Confidence Computing, IN (2024)","DOI":"10.1016\/j.hcc.2024.100211"},{"key":"16_CR4","doi-asserted-by":"crossref","unstructured":"Wu, X., Duan, R., Ni, J.: Unveiling security, privacy, and ethical concerns of ChatGPT. J. Inf. Intell. (2024)","DOI":"10.1016\/j.jiixd.2023.10.007"},{"key":"16_CR5","unstructured":"Wei, A., Haghtalab, N., Steinhardt, J.: Jailbroken: how does LLM safety training fail?. In : Proceedings of the 37th International Conference on Neural Information Processing Systems (2023)"},{"key":"16_CR6","unstructured":"Ouyang, L., et al.: Training language models to follow instructions with human feedback. In : Advances in Neural Information Processing Systems (2022)"},{"key":"16_CR7","doi-asserted-by":"crossref","unstructured":"Zhang, Z., Yang, J., Ke, P., Mi, F., Wang, H., Huang, M.: Defending large language models against jailbreaking attacks through goal prioritization. In: Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics (2023)","DOI":"10.18653\/v1\/2024.acl-long.481"},{"key":"16_CR8","doi-asserted-by":"crossref","unstructured":"Xie, Y., et al.: Defending ChatGPT against jailbreak attack via self-reminders. In : Nature Machine Intelligence (2023)","DOI":"10.1038\/s42256-023-00765-8"},{"key":"16_CR9","unstructured":"Yuan, Y., et al.: GPT-4 is too smart to be safe: stealthy chat with LLMs via Cipher. In: The Twelfth International Conference on Learning Representations (2024)"},{"key":"16_CR10","doi-asserted-by":"crossref","unstructured":"Cao, B., Cao, Y., Lin, L., Chen, J.: Defending against alignment-breaking attacks via robustly aligned LLM. In: Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics (2024)","DOI":"10.18653\/v1\/2024.acl-long.568"},{"key":"16_CR11","unstructured":"Deng, Y., Zhang, W., Pan, S.J., Bing, L.: Multilingual jailbreak challenges in large language models. In: The Twelfth International Conference on Learning Representations (2024)"},{"key":"16_CR12","unstructured":"Liu, X., Xu, N., Chen, M., Xiao, C.: AutoDAN: generating stealthy jailbreak prompts on aligned large language models. In : The Twelfth International Conference on Learning Representations (2023)"},{"key":"16_CR13","unstructured":"Li, X., Zhou, Z., Zhu, J., Yao, J., Liu, T., Han, B.: DeepInception: hypnotize large language model to be jailbreaker. CoRR abs\/2311.03191 (2023)"},{"key":"16_CR14","unstructured":"Zhou, W., et al.: EasyJailbreak: a unified framework for jailbreaking large language models. CoRR abs\/2403.12171 (2024)"},{"key":"16_CR15","doi-asserted-by":"crossref","unstructured":"Shen, X., Chen, Z., Backes, M., Shen, Y., Zhang, Y.: \u201cdo anything now\u201d: characterizing and evaluating in-the-wild jailbreak prompts on large language models. In: Proceedings of the 2024 on ACM SIGSAC Conference on Computer and Communications Security (2024)","DOI":"10.1145\/3658644.3670388"},{"key":"16_CR16","unstructured":"Phute, M., et al.: LLM self defense: by self examination, LLMs know they are being tricked. CoRR 2308.07308 (2023)"},{"key":"16_CR17","unstructured":"Zou, A., Wang, Z., Carlini, N., Nasr, M., Kolter, J. Z., Fredrikson, M.: Universal and transferable adversarial attacks on aligned language models. CORR 2307.15043 (2023)"},{"key":"16_CR18","unstructured":"Wei, Z., Wang, Y., Li, A., Mo, Y., Wang, Y.: Jailbreak and guard aligned language models with only few in-context demonstrations. CORR 2310.06387 (2023)"},{"key":"16_CR19","unstructured":"Shah, R., Pour, S., Tagade, A., Casper, S., Rando, J.: Scalable and transferable black-box jailbreaks for language models via persona modulation. CORR 2311.03348 (2023)"},{"key":"16_CR20","doi-asserted-by":"crossref","unstructured":"Deng, G., et al.: Masterkey: automated jailbreak across multiple large language model chatbots. CORR 2307.08715 (2023)","DOI":"10.14722\/ndss.2024.24188"},{"key":"16_CR21","unstructured":"Robey, A., Wong, E., Hassani, H., Pappas, G. J.: SmoothLLM: defending large language models against jailbreaking attacks. CORR 2310.03684 (2023)"},{"key":"16_CR22","unstructured":"Jain, N., et al.: Baseline defenses for adversarial attacks against aligned language models. CORR 2309.00614 (2023)"},{"key":"16_CR23","doi-asserted-by":"crossref","unstructured":"Wang, M., et al.: Detoxifying large language models via knowledge editing. In: Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics (2024)","DOI":"10.18653\/v1\/2024.acl-long.171"},{"key":"16_CR24","doi-asserted-by":"crossref","unstructured":"Ding, P., et al.: A Wolf in Sheep\u2019s Clothing: generalized nested jailbreak prompts can fool large language models easily. In: Proceedings of the 2024 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies (2024)","DOI":"10.18653\/v1\/2024.naacl-long.118"},{"key":"16_CR25","unstructured":"Liu, Y., et al.: Jailbreaking ChatGPT via prompt engineering: an empirical study. CORR 2305.13860 (2023)"},{"key":"16_CR26","unstructured":"Yu, J., Lin, X., Yu, Z., Xing, X.: GPTFUZZER: red teaming large language models with auto-generated jailbreak prompts. CORR 2309.10253 (2023)"},{"key":"16_CR27","unstructured":"Huang, Y., Gupta, S., Xia, M., Li, K., Chen, D.: Catastrophic jailbreak of open-source LLMs via exploiting generation. In: The Twelfth International Conference on Learning Representations (2024)"},{"key":"16_CR28","unstructured":"Li, X., Zhou, Z., Zhu, J., Yao, J., Liu, T., Han, B.: Deepinception: Hypnotize large language model to be jailbreaker. CORR 2311.03191 (2023)"},{"key":"16_CR29","unstructured":"Xiang, Z., Jiang, F., Xiong, Z., Ramasubramanian, B., Poovendran, R., Li, B.: BadChain: backdoor chain-of-thought prompting for large language models. In: The Twelfth International Conference on Learning Representations (2024)"},{"key":"16_CR30","doi-asserted-by":"crossref","unstructured":"Shaikh, O., Zhang, H., Held, W., Bernstein, M., Yang, D.: On Second Thought, Let\u2019s not think step by step! bias and toxicity in zero-shot reasoning. In: Proceedings of the 61st Annual Meeting of the Association for Computational Linguistics (2023)","DOI":"10.18653\/v1\/2023.acl-long.244"},{"key":"16_CR31","unstructured":"Chu, J., Liu, Y., Yang, Z., Shen, X., Backes, M., Zhang, Y.: Comprehensive assessment of jailbreak attacks against LLMs. CORR 2402.05668 (2024)"},{"key":"16_CR32","unstructured":"Rossi, S., Michel, A.M., Mukkamala, R.R., Thatcher, J.B.: An early categorization of prompt injection attacks on large language models. CORR 2402.00898 (2024)"}],"container-title":["Lecture Notes in Computer Science","Natural Language Processing and Chinese Computing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-95-3352-7_16","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,11,16]],"date-time":"2025-11-16T14:09:07Z","timestamp":1763302147000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-95-3352-7_16"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,11,17]]},"ISBN":["9789819533510","9789819533527"],"references-count":32,"URL":"https:\/\/doi.org\/10.1007\/978-981-95-3352-7_16","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,11,17]]},"assertion":[{"value":"17 November 2025","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"NLPCC","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"CCF International Conference on Natural Language Processing and Chinese Computing","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Urumqi","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"China","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2025","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"7 August 2025","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"9 August 2025","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"14","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"nlpcc2025","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"http:\/\/tcci.ccf.org.cn\/conference\/2025\/index.php","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}