{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,23]],"date-time":"2026-07-23T00:15:38Z","timestamp":1784765738119,"version":"3.55.0"},"reference-count":56,"publisher":"Springer Science and Business Media LLC","issue":"12","license":[{"start":{"date-parts":[[2023,12,12]],"date-time":"2023-12-12T00:00:00Z","timestamp":1702339200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,12,12]],"date-time":"2023-12-12T00:00:00Z","timestamp":1702339200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Nat Mach Intell"],"DOI":"10.1038\/s42256-023-00765-8","type":"journal-article","created":{"date-parts":[[2023,12,12]],"date-time":"2023-12-12T06:03:12Z","timestamp":1702360992000},"page":"1486-1496","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":137,"title":["Defending ChatGPT against jailbreak attack via self-reminders"],"prefix":"10.1038","volume":"5","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-5169-3180","authenticated-orcid":false,"given":"Yueqi","family":"Xie","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jingwei","family":"Yi","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8836-1430","authenticated-orcid":false,"given":"Jiawei","family":"Shao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Justin","family":"Curl","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Lingjuan","family":"Lyu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2199-3948","authenticated-orcid":false,"given":"Qifeng","family":"Chen","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xing","family":"Xie","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9138-1272","authenticated-orcid":false,"given":"Fangzhao","family":"Wu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2023,12,12]]},"reference":[{"key":"765_CR1","unstructured":"OpenAI. ChatGPT. openai.com\/blog\/chatgpt (2022)."},{"key":"765_CR2","unstructured":"Jiao, W., Wang, W., Huang, J.-T., Wang, X. & Tu, Z. Is ChatGPT a good translator? A preliminary study. Preprint at arXiv.org\/2301.08745 (2023)."},{"key":"765_CR3","doi-asserted-by":"publisher","first-page":"1055","DOI":"10.1016\/j.jtha.2023.01.011","volume":"21","author":"E Klang","year":"2023","unstructured":"Klang, E. & Levy-Mendelovich, S. Evaluation of OpenAI\u2019s large language model as a new tool for writing papers in the field of thrombosis and hemostasis. J. Thromb. Haemost. 21, 1055\u20131058 (2023).","journal-title":"J. Thromb. Haemost."},{"key":"765_CR4","doi-asserted-by":"publisher","first-page":"e0000198","DOI":"10.1371\/journal.pdig.0000198","volume":"2","author":"TH Kung","year":"2023","unstructured":"Kung, T. H. et al. Performance of ChatGPT on usmle: potential for AI-assisted medical education using large language models. PLoS Digit. Health 2, e0000198 (2023).","journal-title":"PLoS Digit. Health"},{"key":"765_CR5","unstructured":"Reinventing search with a new AI-powered Microsoft Bing and Edge, your copilot for the web. Microsoft blogs.microsoft.com\/blog\/2023\/02\/07\/reinventing-search-with-a-new-ai-powered-microsoft-bing-and-edge-your-copilot-for-the-web\/ (2023)."},{"key":"765_CR6","unstructured":"Introducing Microsoft 365 copilot \u2013 your copilot for work. Microsoft blogs.microsoft.com\/blog\/2023\/03\/16\/introducing-microsoft-365-copilot-your-copilot-for-work\/ (2023)."},{"key":"765_CR7","doi-asserted-by":"crossref","unstructured":"Much to discuss in AI ethics. Nat. Mach. Intell. 4, 1055\u20131056 (2022).","DOI":"10.1038\/s42256-022-00598-x"},{"key":"765_CR8","unstructured":"Brown, T. et al. Language models are few-shot learners. In Proc. Advances in Neural Information Processing Systems Vol. 33 (eds Larochelle, H. et al.) 1877\u20131901 (Curran, 2020)."},{"key":"765_CR9","first-page":"1\u2013113","volume":"24","author":"A Chowdhery","year":"2023","unstructured":"Chowdhery, A. et al. PaLM: scaling language modeling with pathways. J. Mach. Learn. Res. 24, 1\u2013113 (2023).","journal-title":"J. Mach. Learn. Res."},{"key":"765_CR10","unstructured":"Zhang, S. et al. Opt: Open pre-trained transformer language models. Preprint at https:\/\/arXiv.org\/2205.01068 (2022)."},{"key":"765_CR11","unstructured":"Askell, A. et al. A general language assistant as a laboratory for alignment. Preprint at https:\/\/arXiv.org\/2112.00861 (2021)."},{"key":"765_CR12","unstructured":"Bai, Y. et al. Training a helpful and harmless assistant with reinforcement learning from human feedback. Preprint at https:\/\/arXiv.org\/2204.05862 (2022)."},{"key":"765_CR13","doi-asserted-by":"crossref","unstructured":"Kasirzadeh, A. & Gabriel, I. In conversation with artificial intelligence: aligning language models with human values. Preprint at https:\/\/arXiv.org\/2209.00731 (2022).","DOI":"10.1007\/s13347-023-00606-x"},{"key":"765_CR14","unstructured":"Ouyang, L. et al. Training language models to follow instructions with human feedback. In Proc. Advances in Neural Information Processing Systems Vol. 35 (eds Koyejo, S. et al.) 27730\u201327744 (Curran, 2022); http:\/\/papers.nips.cc\/paper_files\/paper\/2022\/hash\/b1efde53be364a73914f58805a001731-Abstract-Conference.html"},{"key":"765_CR15","unstructured":"GPT-4 system card. OpenAI https:\/\/cdn.openai.com\/papers\/gpt-4-system-card.pdf (2023)."},{"key":"765_CR16","unstructured":"Selvi, J. Exploring prompt injection attacks. NCC Group https:\/\/research.nccgroup.com\/2022\/12\/05\/exploring-prompt-injection-attacks\/ (2022)."},{"key":"765_CR17","unstructured":"Daryanani, L. How to jailbreak ChatGPT. Watcher Guru https:\/\/watcher.guru\/news\/how-to-jailbreak-chatgpt\/ (2023)."},{"key":"765_CR18","unstructured":"Warren, T. These are Microsoft\u2019s Bing AI secret rules and why it says it\u2019s named Sydney. The Verge https:\/\/www.theverge.com\/23599441\/microsoft-bing-ai-sydney-secret-rules\/ (2023)."},{"key":"765_CR19","unstructured":"Albert, A. Jailbreak chat. The Prompt Report https:\/\/www.jailbreakchat.com\/ (2023)."},{"key":"765_CR20","unstructured":"ChatGPT \u2013 The Impact of Large Language Models on Law Enforcement (Europol, 2023)."},{"key":"765_CR21","unstructured":"Mitchell, E., Lee, Y., Khazatsky, A., Manning, C. D. & Finn, C. DetectGPT: zero-shot machine-generated text detection using probability curvature. In Proc. International Conference on Machine Learning, ICML 2023 (eds Krause, A. et al.) 24950\u201324962 (PMLR, 2023); https:\/\/proceedings.mlr.press\/v202\/mitchell23a.html"},{"key":"765_CR22","doi-asserted-by":"publisher","first-page":"1166120","DOI":"10.3389\/fpubh.2023.1166120","volume":"11","author":"L De Angelis","year":"2023","unstructured":"De Angelis, L. et al. ChatGPT and the rise of large language models: the new AI-driven infodemic threat in public health. Front. Public Health 11, 1166120 (2023).","journal-title":"Front. Public Health"},{"key":"765_CR23","unstructured":"Dasgupta, I. et al. Language models show human-like content effects on reasoning. Preprint at https:\/\/arXiv.org\/2207.07051 (2022)."},{"key":"765_CR24","unstructured":"Wei, J. et al. Chain-of-thought prompting elicits reasoning in large language models. In Proc. Advances in Neural Information Processing Systems Vol. 35 (eds Koyejo, S. et al.) 24824\u201324837 (Curran, 2022); http:\/\/papers.nips.cc\/paper_files\/paper\/2022\/hash\/9d5609613524ecf4f15af0f7b31abca4-Abstract-Conference.html"},{"key":"765_CR25","unstructured":"Wang, X. et al. Self-consistency improves chain of thought reasoning in language models. In Proc. 11th International Conference on Learning Representations, ICLR 2023 (OpenReview.net, 2023); https:\/\/openreview.net\/pdf?id=1PL1NIMMrw"},{"key":"765_CR26","unstructured":"Zhou, D. et al. Least-to-most prompting enables complex reasoning in large language models. In Proc. 11th International Conference on Learning Representations, ICLR 2023 (OpenReview.net, 2023); https:\/\/openreview.net\/pdf?id=WZH7099tgfM"},{"key":"765_CR27","doi-asserted-by":"publisher","first-page":"493\u2013503","DOI":"10.1037\/0003-066X.54.7.493","volume":"54","author":"PM Gollwitzer","year":"1999","unstructured":"Gollwitzer, P. M. Implementation intentions: strong effects of simple plans. Am. Psychol. 54, 493\u2013503 (1999).","journal-title":"Am. Psychol."},{"key":"765_CR28","unstructured":"Carver, C. S. & Scheier, M. F. On the Self-Regulation of Behavior (Cambridge Univ. Press, 2001)."},{"key":"765_CR29","first-page":"185","volume":"6","author":"D Meichenbaum","year":"1977","unstructured":"Meichenbaum, D. Cognitive behaviour modification. Cogn. Behav. Ther. 6, 185\u2013192 (1977).","journal-title":"Cogn. Behav. Ther."},{"key":"765_CR30","doi-asserted-by":"publisher","first-page":"191\u2013215","DOI":"10.1037\/0033-295X.84.2.191","volume":"84","author":"A Bandura","year":"1977","unstructured":"Bandura, A. Self-efficacy: toward a unifying theory of behavioral change. Psychol. Rev. 84, 191\u2013215 (1977).","journal-title":"Psychol. Rev."},{"key":"765_CR31","unstructured":"Ganguli, D. et al. The capacity for moral self-correction in large language models. Preprint at https:\/\/arXiv.org\/2302.07459 (2023)."},{"key":"765_CR32","unstructured":"Kadavath, S. et al. Language models (mostly) know what they know. Preprint at https:\/\/arXiv.org\/2207.05221 (2022)."},{"key":"765_CR33","doi-asserted-by":"publisher","first-page":"1408","DOI":"10.1162\/tacl_a_00434","volume":"9","author":"T Schick","year":"2021","unstructured":"Schick, T., Udupa, S. & Sch\u00fctze, H. Self-diagnosis and self-debiasing: a proposal for reducing corpus-based bias in NLP. Trans. Assoc. Comput. Linguist. 9, 1408\u20131424 (2021).","journal-title":"Trans. Assoc. Comput. Linguist."},{"key":"765_CR34","unstructured":"Touvron, H. et al. Llama: open and efficient foundation language models. Preprint at https:\/\/arXiv.org\/2302.13971 (2023)."},{"key":"765_CR35","unstructured":"Touvron, H. et al. Llama 2: open foundation and fine-tuned chat models. Preprint at https:\/\/arXiv.org\/2307.09288 (2023)."},{"key":"765_CR36","unstructured":"Wang, A. et al. GLUE: a multi-task benchmark and analysis platform for natural language understanding. In Proc. 7th International Conference on Learning Representations, ICLR 2019 (OpenReview.net, 2019); https:\/\/openreview.net\/forum?id=rJ4km2R5t7"},{"key":"765_CR37","unstructured":"Shi, F. et al. Language models are multilingual chain-of-thought reasoners. In Proc. 11th International Conference on Learning Representations, ICLR 2023 (OpenReview.net, 2023); https:\/\/openreview.net\/pdf?id=fR3wGCk-IXp"},{"key":"765_CR38","doi-asserted-by":"crossref","unstructured":"See, A., Liu, P. J. & Manning, C. D. Get to the point: summarization with pointer-generator networks. In Proc. 55th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers) (eds Barzilay, R. & Kan, M.-Y.), 1073\u20131083 (Association for Computational Linguistics, 2017); https:\/\/www.aclweb.org\/anthology\/P17-1099","DOI":"10.18653\/v1\/P17-1099"},{"key":"765_CR39","doi-asserted-by":"publisher","unstructured":"Narayan, S., Cohen, S. B. & Lapata, M. Don\u2019t give me the details, just the summary! Topic-aware convolutional neural networks for extreme summarization. In Proc. 2018 Conference on Empirical Methods in Natural Language Processing (eds Riloff, E. et al.) 1797\u20131807 (Association for Computational Linguistics, 2018); https:\/\/doi.org\/10.18653\/v1\/d18-1206","DOI":"10.18653\/v1\/d18-1206"},{"key":"765_CR40","unstructured":"Kasai, J., Pappas, N., Peng, H., Cross, J. & Smith, N. A. Deep encoder, shallow decoder: reevaluating non-autoregressive machine translation. In Proc. 9th International Conference on Learning Representations, ICLR 2021 (OpenReview.net, 2021); https:\/\/openreview.net\/forum?id=KpfasTaLUpq"},{"key":"765_CR41","doi-asserted-by":"publisher","unstructured":"Rajpurkar, P., Zhang, J., Lopyrev, K. & Liang, P. Squad: 100,000+ questions for machine comprehension of text. In Proc. 2016 Conference on Empirical Methods in Natural Language Processing (eds Su, J. et al.) 2383\u20132392 (Association for Computational Linguistics, 2016); https:\/\/doi.org\/10.18653\/v1\/d16-1264","DOI":"10.18653\/v1\/d16-1264"},{"key":"765_CR42","doi-asserted-by":"publisher","first-page":"319","DOI":"10.1007\/s11218-011-9152-4","volume":"14","author":"RJ Harnish","year":"2011","unstructured":"Harnish, R. J. & Bridges, K. R. Effect of syllabus tone: students\u2019 perceptions of instructor and course. Soc. Psychol. Educ. 14, 319\u2013330 (2011).","journal-title":"Soc. Psychol. Educ."},{"key":"765_CR43","doi-asserted-by":"publisher","first-page":"139","DOI":"10.1901\/jaba.1968.1-139","volume":"1","author":"CH Madsen Jr","year":"1968","unstructured":"Madsen Jr, C. H., Becker, W. C. & Thomas, D. R. Rules, praise, and ignoring: elements of elementary classroom control 1. J. Appl. Behav. Anal. 1, 139\u2013150 (1968).","journal-title":"J. Appl. Behav. Anal."},{"key":"765_CR44","doi-asserted-by":"crossref","unstructured":"Li, H., Guo, D., Fan, W., Xu, M. & Song, Y. Multi-step jailbreaking privacy attacks on ChatGPT. Preprint at https:\/\/arXiv.org\/2304.05197 (2023).","DOI":"10.18653\/v1\/2023.findings-emnlp.272"},{"key":"765_CR45","doi-asserted-by":"crossref","unstructured":"Klimt, B. & Yang, Y. The Enron corpus: a new dataset for email classification research. In European Conference on Machine Learning (eds Boulicaut, J. F. et al.) 217\u2013226 (Springer, 2004).","DOI":"10.1007\/978-3-540-30115-8_22"},{"key":"765_CR46","doi-asserted-by":"crossref","unstructured":"Pryzant, R. et al. Automatic prompt optimization with \u2018gradient descent\u2019 and beam search. Preprint at https:\/\/arXiv.org\/2305.03495 (2023).","DOI":"10.18653\/v1\/2023.emnlp-main.494"},{"key":"765_CR47","unstructured":"Bubeck, S. et al. Sparks of artificial general intelligence: early experiments with GPT-4. Preprint at https:\/\/arXiv.org\/2303.12712 (2023)."},{"key":"765_CR48","unstructured":"Let\u2019s chat about ChatGPT. UBS https:\/\/www.ubs.com\/global\/en\/wealth-management\/our-approach\/marketnews\/article.1585717.html (2023)."},{"key":"765_CR49","unstructured":"Perez, F. & Ribeiro, I. Ignore previous prompt: attack techniques for language models. Preprint at https:\/\/arXiv.org\/2211.09527 (2022)."},{"key":"765_CR50","unstructured":"Greshake, K. et al. More than you\u2019ve asked for: a comprehensive analysis of novel prompt injection threats to application-integrated large language models. Preprint at https:\/\/arXiv.org\/2302.12173 (2023)."},{"key":"765_CR51","unstructured":"Liu, Y. et al. Jailbreaking ChatGPT via prompt engineering: an empirical study. Preprint at https:\/\/arXiv.org\/2305.13860 (2023)."},{"key":"765_CR52","unstructured":"Shen, X., Chen, Z., Backes, M., Shen, Y. & Zhang, Y. \u2018Do anything now\u2019: characterizing and evaluating in-the-wild jailbreak prompts on large language models. Preprint at https:\/\/arXiv.org\/2308.03825 (2023)."},{"key":"765_CR53","unstructured":"Zhang, T., Liu, F., Wong, J., Abbeel, P. & Gonzalez, J. E. The wisdom of hindsight makes language models better instruction followers. In Proc. International Conference on Machine Learning, ICML 2023 (eds Krause, A. et al.) 41414\u201341428 (PMLR, 2023); https:\/\/proceedings.mlr.press\/v202\/zhang23ab.html"},{"key":"765_CR54","unstructured":"Devlin, J., Chang, M.-W., Lee, K. & Toutanova, K. Bert: pre-training of deep bidirectional transformers for language understanding. In Proc. 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, Volume 1 (Long and Short Papers) (eds Burstein, J. et al.) 4171\u20134186 (Association for Computational Linguistics, 2019)."},{"key":"765_CR55","doi-asserted-by":"publisher","unstructured":"Yi, J. yjw1029\/self-reminder-data: v.1.0.0 (Zenodo, 2023); https:\/\/doi.org\/10.5281\/zenodo.10043052","DOI":"10.5281\/zenodo.10043052"},{"key":"765_CR56","doi-asserted-by":"publisher","unstructured":"Yi, J. yjw1029\/self-reminder: v.1.0.0 (Zenodo, 2023); https:\/\/doi.org\/10.5281\/zenodo.10043044","DOI":"10.5281\/zenodo.10043044"}],"container-title":["Nature Machine Intelligence"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/www.nature.com\/articles\/s42256-023-00765-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/www.nature.com\/articles\/s42256-023-00765-8","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/www.nature.com\/articles\/s42256-023-00765-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,12,18]],"date-time":"2023-12-18T15:09:41Z","timestamp":1702912181000},"score":1,"resource":{"primary":{"URL":"https:\/\/www.nature.com\/articles\/s42256-023-00765-8"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,12,12]]},"references-count":56,"journal-issue":{"issue":"12","published-online":{"date-parts":[[2023,12]]}},"alternative-id":["765"],"URL":"https:\/\/doi.org\/10.1038\/s42256-023-00765-8","relation":{"has-preprint":[{"id-type":"doi","id":"10.21203\/rs.3.rs-2873090\/v1","asserted-by":"object"}]},"ISSN":["2522-5839"],"issn-type":[{"value":"2522-5839","type":"electronic"}],"subject":[],"published":{"date-parts":[[2023,12,12]]},"assertion":[{"value":"19 May 2023","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"28 October 2023","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"12 December 2023","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"The authors declare no competing interests.","order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Competing interests"}}]}}