{
  "id": "instruction-hierarchy",
  "code": "PTL-0093",
  "term": "Instruction Hierarchy",
  "aliases": [],
  "category": "security",
  "definition": "The instruction hierarchy is a training approach that teaches a model to prioritize instructions by source, typically system over user over tool output, and to ignore lower-priority instructions that conflict with higher-priority ones.",
  "description": "Wallace et al. trained models on synthetic conflicts and found large gains in robustness to prompt injection and system-prompt extraction with limited loss in helpfulness.",
  "example": null,
  "broader": [],
  "narrower": [],
  "related": [
    "system-prompt",
    "prompt-injection",
    "spotlighting"
  ],
  "introduced": 2024,
  "sources": [
    {
      "title": "The Instruction Hierarchy: Training LLMs to Prioritize Privileged Instructions",
      "authors": "Wallace et al.",
      "year": 2024,
      "url": "https://arxiv.org/abs/2404.13208"
    }
  ],
  "url": "https://protologue.com/t/instruction-hierarchy/",
  "citation": "Protologue. (2026). Instruction Hierarchy. In Protologue: A Taxonomy of Prompting and LLM Techniques (v1.0.0, PTL-0093). https://protologue.com/t/instruction-hierarchy/"
}