{
  "id": "test-time-compute-scaling",
  "code": "PTL-0084",
  "term": "Test-Time Compute Scaling",
  "aliases": [
    "inference-time scaling",
    "test-time scaling"
  ],
  "category": "test-time",
  "definition": "Test-time compute scaling improves a model's answers by spending more computation at inference, through longer reasoning, more samples, search, or verification, rather than by training a larger model.",
  "description": "Snell et al. found that allocating test-time compute adaptively per prompt could be more effective than scaling model parameters for some problems. Self-consistency, best-of-N, tree search, and reasoning models are all forms of test-time scaling.",
  "example": null,
  "broader": [],
  "narrower": [],
  "related": [
    "reasoning-model",
    "self-consistency",
    "best-of-n-sampling",
    "tree-of-thoughts",
    "process-reward-model"
  ],
  "introduced": 2024,
  "sources": [
    {
      "title": "Scaling LLM Test-Time Compute Optimally can be More Effective than Scaling Model Parameters",
      "authors": "Snell et al.",
      "year": 2024,
      "url": "https://arxiv.org/abs/2408.03314"
    }
  ],
  "url": "https://protologue.com/t/test-time-compute-scaling/",
  "citation": "Protologue. (2026). Test-Time Compute Scaling. In Protologue: A Taxonomy of Prompting and LLM Techniques (v1.0.0, PTL-0084). https://protologue.com/t/test-time-compute-scaling/"
}