{
  "id": "jailbreak",
  "code": "PTL-0090",
  "term": "Jailbreak",
  "aliases": [
    "jailbreaking"
  ],
  "category": "security",
  "definition": "A jailbreak is a prompt crafted to make a model produce outputs its safety training is meant to prevent, often through role-play, hypothetical framing, obfuscation, or other adversarial techniques.",
  "description": "Wei et al. attributed jailbreak success to two failure modes, competing objectives between helpfulness and safety, and mismatched generalization, where safety training does not cover inputs the model can still understand. Jailbreaks target the model's safety behavior, while prompt injection targets the application's instructions.",
  "example": null,
  "broader": [],
  "narrower": [
    "adversarial-suffix",
    "many-shot-jailbreaking"
  ],
  "related": [
    "prompt-injection",
    "constitutional-ai"
  ],
  "introduced": null,
  "sources": [
    {
      "title": "Jailbroken: How Does LLM Safety Training Fail?",
      "authors": "Wei et al.",
      "year": 2023,
      "url": "https://arxiv.org/abs/2307.02483"
    }
  ],
  "url": "https://protologue.com/t/jailbreak/",
  "citation": "Protologue. (2026). Jailbreak. In Protologue: A Taxonomy of Prompting and LLM Techniques (v1.0.0, PTL-0090). https://protologue.com/t/jailbreak/"
}