{
  "id": "token",
  "code": "PTL-0005",
  "term": "Token",
  "aliases": [
    "subword",
    "BPE token"
  ],
  "category": "foundations",
  "definition": "A token is the basic unit of text a language model reads and writes, typically a word, word fragment, or character sequence produced by a subword tokenizer such as byte-pair encoding.",
  "description": "Context limits, pricing, and generation speed are all measured in tokens. Because tokenization splits text unevenly, character-level tasks such as counting letters or reversing strings are harder for models than they appear, and the same content can cost different numbers of tokens in different languages.",
  "example": null,
  "broader": [],
  "narrower": [],
  "related": [
    "context-window"
  ],
  "introduced": null,
  "sources": [
    {
      "title": "Neural Machine Translation of Rare Words with Subword Units",
      "authors": "Sennrich et al.",
      "year": 2015,
      "url": "https://arxiv.org/abs/1508.07909"
    }
  ],
  "url": "https://protologue.com/t/token/",
  "citation": "Protologue. (2026). Token. In Protologue: A Taxonomy of Prompting and LLM Techniques (v1.0.0, PTL-0005). https://protologue.com/t/token/"
}