{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,4]],"date-time":"2026-08-04T05:43:48Z","timestamp":1785822228612,"version":"3.56.0"},"reference-count":112,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"8","license":[{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Artif. Intell."],"published-print":{"date-parts":[[2026,8]]},"DOI":"10.1109\/tai.2026.3665656","type":"journal-article","created":{"date-parts":[[2026,2,17]],"date-time":"2026-02-17T21:11:38Z","timestamp":1771362698000},"page":"4237-4251","source":"Crossref","is-referenced-by-count":2,"title":["Prompt-Based Jailbreaking of Leading LLM Chatbots: A Survey of Attacks and Defenses"],"prefix":"10.1109","volume":"7","author":[{"ORCID":"https:\/\/orcid.org\/0009-0006-8623-6592","authenticated-orcid":false,"given":"Brynn","family":"Knowlton","sequence":"first","affiliation":[{"name":"California State University San Bernardino (CSUSB) San Bernardino","place":["San Bernardino, CA, USA"],"department":["School of Computer Science and Engineering"]}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-5227-5885","authenticated-orcid":false,"given":"Jovani","family":"Campa","sequence":"additional","affiliation":[{"name":"California State University San Bernardino (CSUSB) San Bernardino","place":["San Bernardino, CA, USA"],"department":["School of Computer Science and Engineering"]}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-0914-6927","authenticated-orcid":false,"given":"David Solis","family":"Gallo","sequence":"additional","affiliation":[{"name":"California State University San Bernardino (CSUSB) San Bernardino","place":["San Bernardino, CA, USA"],"department":["School of Computer Science and Engineering"]}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1061-4236","authenticated-orcid":false,"given":"Khalil","family":"Dajani","sequence":"additional","affiliation":[{"name":"California State University San Bernardino (CSUSB) San Bernardino","place":["San Bernardino, CA, USA"],"department":["School of Computer Science and Engineering"]}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-5671-7277","authenticated-orcid":false,"given":"Nabeel","family":"Alzahrani","sequence":"additional","affiliation":[{"name":"California State University San Bernardino (CSUSB) San Bernardino","place":["San Bernardino, CA, USA"],"department":["School of Computer Science and Engineering"]}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","article-title":"ChatGPT","year":"2025"},{"key":"ref2","article-title":"Claude AI","year":"2025"},{"key":"ref3","article-title":"Gemini","year":"2025"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/CCWC62904.2025.10903781"},{"key":"ref5","article-title":"AutoJailbreak: Exploring jailbreak attacks and defenses through a dependency lens","author":"Lu","year":"2024"},{"key":"ref6","article-title":"Adversarial suffix filtering: A defense pipeline for LLMs","author":"Khachaturov","year":"2025"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/ISNCC62547.2024.10758949"},{"key":"ref8","article-title":"LLMStinger: Jailbreaking LLMs using RL fine-tuned LLMs","author":"Jha","year":"2024"},{"key":"ref9","article-title":"PromptBench: A unified library for evaluation of large language models","author":"Zhu","year":"2024"},{"key":"ref10","doi-asserted-by":"crossref","DOI":"10.52202\/079017-1745","article-title":"JailbreakBench: An open robustness benchmark for jailbreaking large language models","author":"Chao","year":"2024"},{"key":"ref11","article-title":"IEEE Xplore digital library","year":"2025"},{"key":"ref12","article-title":"arxiv.org e-print archive","year":"2025"},{"key":"ref13","article-title":"ACM digital library","year":"2025"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.naacl-long.375"},{"key":"ref15","article-title":"Jailbreak attacks and defenses against large language models: A survey","author":"Yi","year":"2024"},{"key":"ref16","doi-asserted-by":"crossref","DOI":"10.52202\/079017-1270","article-title":"Robust prompt optimization for defending language models against jailbreaking attacks","author":"Zhou","year":"2024"},{"key":"ref17","article-title":"SneakyPrompt: Jailbreaking text-to-image generative models","author":"Yang","year":"2023"},{"key":"ref18","article-title":"GhostPrompt: Jailbreaking text-to-image generative models based on dynamic optimization","author":"Chen","year":"2025"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1145\/3658644.3670370"},{"key":"ref20","article-title":"BitBypass: A new direction in jailbreaking aligned large language models with bitstream camouflage","author":"Nakka","year":"2025"},{"key":"ref21","doi-asserted-by":"crossref","DOI":"10.18653\/v1\/2025.findings-emnlp.114","article-title":"Exploring the vulnerability of the content moderation guardrail in large language models via intent manipulation","author":"Zhuang","year":"2025"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1145\/3689932.3694764"},{"key":"ref23","doi-asserted-by":"crossref","DOI":"10.52202\/079017-1210","article-title":"Mission impossible: A statistical perspective on jailbreaking LLMs","author":"Su","year":"2024"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1145\/3689217.3690618"},{"key":"ref25","doi-asserted-by":"crossref","DOI":"10.18653\/v1\/2024.acl-long.773","article-title":"How Johnny can persuade LLMs to jailbreak them: Rethinking persuasion to challenge AI safety by humanizing LLMs","author":"Zeng","year":"2024"},{"key":"ref26","article-title":"Generation, detection, and evaluation of role-play based jailbreak attacks in large language models","author":"Johnson","year":"2024"},{"key":"ref27","article-title":"Jailbreak defense in a narrow domain: Limitations of existing methods and a new transcript-classifier approach","author":"Wang","year":"2024"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/BigData62323.2024.10825802"},{"key":"ref29","article-title":"Adaptive jailbreaking strategies based on the semantic understanding capabilities of large language models","author":"Yu","year":"2025"},{"key":"ref30","article-title":"Unmasking the canvas: A dynamic benchmark for image generation jailbreaking and LLM content safety","author":"Nair","year":"2025"},{"key":"ref31","article-title":"Universal and transferable adversarial attacks on aligned language models","author":"Zou","year":"2023"},{"key":"ref32","article-title":"AutoDefense: Multi-agent LLM defense against jailbreak attacks","author":"Zeng","year":"2024"},{"key":"ref33","article-title":"Best-of-N jailbreaking","author":"Hughes","year":"2024"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1109\/JIOT.2025.3525741"},{"key":"ref35","doi-asserted-by":"crossref","DOI":"10.18653\/v1\/2024.findings-acl.61","article-title":"Realistic evaluation of toxicity in large language models","author":"Luong","year":"2024"},{"key":"ref36","article-title":"Leveraging the potential of prompt engineering for hate speech detection in low-resource languages","author":"Prome","year":"2025"},{"key":"ref37","article-title":"Imprompter: Tricking LLM agents into improper tool use","author":"Fu","year":"2024"},{"key":"ref38","article-title":"Red-teaming large language models using chain of utterances for safety-alignment","author":"Bhardwaj","year":"2023"},{"key":"ref39","article-title":"AutoDAN: Interpretable gradient-based adversarial attacks on large language models","author":"Zhu","year":"2023"},{"issue":"9","key":"ref40","doi-asserted-by":"crossref","DOI":"10.3390\/app14093558","article-title":"All in how you ask for it: Simple black-box method for jailbreak attacks","volume":"14","author":"Takemoto","year":"2024","journal-title":"Appl. Sci."},{"key":"ref41","doi-asserted-by":"crossref","DOI":"10.18653\/v1\/2025.findings-acl.410","article-title":"Breaking the ceiling: Exploring the potential of jailbreak attacks through expanding strategy space","author":"Huang","year":"2025"},{"key":"ref42","article-title":"Great, now write an article about that: The crescendo multi-turn LLM jailbreak attack","author":"Russinovich","year":"2025"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1145\/3717067"},{"key":"ref44","doi-asserted-by":"crossref","DOI":"10.18653\/v1\/2024.naacl-long.118","article-title":"A wolf in sheep\u2019s clothing: Generalized nested jailbreak prompts can fool large language models easily","author":"Ding","year":"2024"},{"key":"ref45","article-title":"Low-resource languages jailbreak GPT-4","author":"Yong","year":"2024"},{"key":"ref46","article-title":"Visual adversarial examples jailbreak aligned large language models","author":"Qi","year":"2023"},{"key":"ref47","article-title":"\u2018Do as I say not as I do\u2019: A semi-automated approach for jailbreak prompt attack against multimodal LLMs","author":"Chiu","year":"2025"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.1109\/icassp48485.2024.10448041"},{"key":"ref49","doi-asserted-by":"crossref","DOI":"10.18653\/v1\/2025.acl-long.101","article-title":"What really matters in many-shot attacks? An empirical study of long-context vulnerabilities in LLMs","author":"Kim","year":"2025"},{"key":"ref50","article-title":"Steering dialogue dynamics for robustness against multi-turn jailbreaking attacks","author":"Hu","year":"2025"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.1109\/TPS-ISA62245.2024.00036"},{"key":"ref52","article-title":"GUARD: Role-playing to generate natural-language jailbreakings to test guideline adherence of large language models","author":"Jin","year":"2025"},{"key":"ref53","article-title":"Rethinking how to evaluate language model jailbreak","author":"Cai","year":"2024"},{"key":"ref54","article-title":"A red teaming roadmap towards system-level safety","author":"Wang","year":"2025"},{"key":"ref55","doi-asserted-by":"publisher","DOI":"10.1145\/3706598.3714025"},{"key":"ref56","article-title":"JailGuard: A universal detection framework for LLM prompt-based attacks","author":"Zhang","year":"2025"},{"key":"ref57","article-title":"Explore, establish, exploit: Red teaming language models from scratch","author":"Casper","year":"2023"},{"key":"ref58","doi-asserted-by":"crossref","DOI":"10.1109\/ICAIRC64177.2024.10900215","article-title":"Robustness of large language models against adversarial attacks","author":"Tao","year":"2024"},{"key":"ref59","article-title":"RL-JACK: Reinforcement learning-powered black-box jailbreaking attack against LLMs","author":"Chen","year":"2024"},{"key":"ref60","article-title":"Systematically analyzing prompt injection vulnerabilities in diverse LLM architectures","author":"Benjamin","year":"2024"},{"key":"ref61","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.trustnlp-1.18"},{"key":"ref62","article-title":"DecodingTrust: A comprehensive assessment of trustworthiness in GPT models","author":"Wang","year":"2024"},{"key":"ref63","article-title":"Optimization-based prompt injection attack to LLM-as-a-judge","author":"Shi","year":"2025"},{"key":"ref64","article-title":"InfoFlood: Jailbreaking large language models with information overload","author":"Yadav","year":"2025"},{"key":"ref65","article-title":"Jailbreaking LLMs\u2019 safeguard with universal magic words for text embedding models","author":"Liang","year":"2025"},{"key":"ref66","article-title":"Jailbreaking large language models with symbolic mathematics","author":"Bethany","year":"2024"},{"key":"ref67","article-title":"Automatic jailbreaking of the text-to-image generative AI systems","author":"Kim","year":"2024"},{"key":"ref68","doi-asserted-by":"crossref","DOI":"10.52202\/079017-1952","article-title":"Tree of attacks: Jailbreaking black-box LLMs automatically","author":"Mehrotra","year":"2024"},{"key":"ref69","doi-asserted-by":"publisher","DOI":"10.1145\/3696410.3714654"},{"key":"ref70","article-title":"XBreaking: Explainable artificial intelligence for jailbreaking LLMs","author":"Arazzi","year":"2025"},{"key":"ref71","article-title":"Jailbreaking black box large language models in twenty queries","author":"Chao","year":"2024"},{"key":"ref72","doi-asserted-by":"publisher","DOI":"10.14722\/ndss.2024.24188"},{"key":"ref73","doi-asserted-by":"crossref","DOI":"10.52202\/075280-3508","article-title":"Jailbroken: How does LLM safety training fail?","author":"Wei","year":"2023"},{"key":"ref74","article-title":"Constitutional classifiers: Defending against universal jailbreaks across thousands of hours of red teaming","author":"Sharma","year":"2025"},{"key":"ref75","doi-asserted-by":"crossref","DOI":"10.18653\/v1\/2024.acl-long.303","article-title":"SafeDecoding: Defending against jailbreak attacks via safety-aware decoding","author":"Xu","year":"2024"},{"key":"ref76","article-title":"Improved large language model jailbreak detection via pretrained embeddings","author":"Galinkin","year":"2024"},{"key":"ref77","article-title":"Jailbreakhunter: A visual analytics approach for jailbreak prompts discovery from large-scale human-LLM conversational datasets","author":"Jin","year":"2024"},{"key":"ref78","article-title":"Emoji attack: Enhancing jailbreak attacks against judge LLM detection","author":"Wei","year":"2025"},{"key":"ref79","article-title":"Aligning large multimodal models with factually augmented RLHF","author":"Sun","year":"2023"},{"key":"ref80","doi-asserted-by":"crossref","DOI":"10.18653\/v1\/2024.findings-emnlp.139","article-title":"How alignment and jailbreak work: Explain LLM safety through intermediate hidden states","author":"Zhou","year":"2024"},{"key":"ref81","article-title":"Cognitive overload attack: Prompt injection for long context","author":"Upadhayay","year":"2024"},{"key":"ref82","article-title":"Certifying LLM safety against adversarial prompting","author":"Kumar","year":"2025"},{"key":"ref83","article-title":"Break the breakout: Reinventing lm defense against jailbreak attacks with self-refinement","author":"Kim","year":"2024"},{"key":"ref84","article-title":"Refusal-trained LLMs are easily jailbroken as browser agents","author":"Kumar","year":"2024"},{"key":"ref85","article-title":"AgentHarm: A benchmark for measuring harmfulness of LLM agents","author":"Andriushchenko","year":"2025"},{"key":"ref86","article-title":"Jailbreaking GPT-4V via self-adversarial attacks with system prompts","author":"Wu","year":"2024"},{"key":"ref87","doi-asserted-by":"publisher","DOI":"10.1145\/3719027.3744836"},{"key":"ref88","doi-asserted-by":"crossref","DOI":"10.1609\/aies.v7i1.31638","article-title":"MoJE: Mixture of jailbreak experts, naive tabular classifiers as guard for prompt attacks","author":"Cornacchia","year":"2024"},{"key":"ref89","doi-asserted-by":"crossref","DOI":"10.1109\/BigData62323.2024.10825103","article-title":"SoK: Prompt hacking of large language models","author":"Rababah","year":"2024"},{"key":"ref90","article-title":"Jailbreaking ChatGPT via prompt engineering: An empirical study","author":"Liu","year":"2024"},{"key":"ref91","article-title":"LARGO: Latent adversarial reflection through gradient optimization for jailbreaking LLMs","author":"Li","year":"2025"},{"key":"ref92","doi-asserted-by":"crossref","DOI":"10.18653\/v1\/2024.findings-acl.948","article-title":"Defending LLMs against jailbreaking attacks via backtranslation","author":"Wang","year":"2024"},{"key":"ref93","doi-asserted-by":"crossref","DOI":"10.1145\/3696410.3714632","article-title":"You can\u2019t eat your cake and have it too: The performance degradation of LLMs with jailbreak defense","author":"Mai","year":"2025"},{"key":"ref94","doi-asserted-by":"publisher","DOI":"10.1109\/ICCA62237.2024.10928071"},{"key":"ref95","article-title":"Red teaming the mind of the machine: A systematic evaluation of prompt injection and jailbreak vulnerabilities in LLMs","author":"Pathade","year":"2025"},{"key":"ref96","article-title":"Jailbreakv: A benchmark for assessing the robustness of multimodal large language models against jailbreak attacks","author":"Luo","year":"2024"},{"key":"ref97","doi-asserted-by":"crossref","DOI":"10.1109\/CVPR52734.2025.02786","article-title":"Playing the fool: Jailbreaking LLMs and multimodal LLMs with out-of-distribution strategy","author":"Jeong","year":"2025"},{"key":"ref98","doi-asserted-by":"crossref","DOI":"10.18653\/v1\/2025.findings-naacl.172","article-title":"Jailbreaking prompt attack: A controllable adversarial attack against diffusion models","author":"Ma","year":"2025"},{"key":"ref99","article-title":"CARES: Comprehensive evaluation of safety and adversarial robustness in medical LLMs","author":"Chen","year":"2025"},{"key":"ref100","article-title":"MMed-RAG: Versatile multimodal rag system for medical vision language models","author":"Xia","year":"2025"},{"key":"ref101","doi-asserted-by":"publisher","DOI":"10.1561\/3300000041"},{"key":"ref102","article-title":"Towards safe AI clinicians: A comprehensive study on large language model jailbreaking in healthcare","author":"Zhang","year":"2025"},{"key":"ref103","article-title":"A jailbroken GenAI model can cause substantial harm: GenAI-powered applications are vulnerable to PromptWares","author":"Cohen","year":"2024"},{"key":"ref104","article-title":"CEE: An inference-time jailbreak defense for embodied intelligence via subspace concept rotation","author":"Yang","year":"2025"},{"key":"ref105","article-title":"Token highlighter: Inspecting and mitigating jailbreak prompts for large language models","author":"Hu","year":"2024"},{"key":"ref106","article-title":"Jailbreaking LLM-controlled robots","author":"Robey","year":"2024"},{"key":"ref107","doi-asserted-by":"crossref","DOI":"10.1109\/ISDFS60797.2024.10527300","article-title":"AbuseGPT: Abuse of generative AI ChatBots to create smishing campaigns","author":"Shibli","year":"2024"},{"key":"ref108","article-title":"Controllable safety alignment: Inference-time adaptation to diverse safety requirements","author":"Zhang","year":"2025"},{"key":"ref109","doi-asserted-by":"crossref","DOI":"10.18653\/v1\/2024.acl-long.481","article-title":"Defending large language models against jailbreaking attacks through goal prioritization","author":"Zhang","year":"2024"},{"key":"ref110","article-title":"WordGame: Efficient & effective LLM jailbreak via simultaneous obfuscation in query and response","author":"Zhang","year":"2024"},{"key":"ref111","article-title":"Catastrophic jailbreak of open-source LLMs via exploiting generation","author":"Huang","year":"2023"},{"key":"ref112","article-title":"Don\u2019t listen to me: Understanding and exploring jailbreak prompts of large language models","author":"Yu","year":"2024"}],"container-title":["IEEE Transactions on Artificial Intelligence"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/9078688\/11635975\/11397677.pdf?arnumber=11397677","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,8,4]],"date-time":"2026-08-04T04:45:06Z","timestamp":1785818706000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11397677\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,8]]},"references-count":112,"journal-issue":{"issue":"8"},"URL":"https:\/\/doi.org\/10.1109\/tai.2026.3665656","relation":{},"ISSN":["2691-4581"],"issn-type":[{"value":"2691-4581","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,8]]}}}