{
  "id": "llm-as-a-judge",
  "code": "PTL-0050",
  "term": "LLM-as-a-Judge",
  "aliases": [
    "model-graded evaluation",
    "LLM evaluator",
    "autorater"
  ],
  "category": "verification",
  "definition": "LLM-as-a-judge is the use of a strong language model to grade, score, or compare the outputs of models against criteria, as a scalable substitute for human evaluation.",
  "description": "Zheng et al. found strong model judges agreed with human preferences at rates comparable to agreement between humans, while documenting biases toward the first-listed answer, longer answers, and the judge's own outputs.",
  "example": null,
  "broader": [],
  "narrower": [],
  "related": [
    "universal-self-consistency",
    "evaluator-optimizer",
    "process-reward-model"
  ],
  "introduced": 2023,
  "sources": [
    {
      "title": "Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena",
      "authors": "Zheng et al.",
      "year": 2023,
      "url": "https://arxiv.org/abs/2306.05685"
    }
  ],
  "url": "https://protologue.com/t/llm-as-a-judge/",
  "citation": "Protologue. (2026). LLM-as-a-Judge. In Protologue: A Taxonomy of Prompting and LLM Techniques (v1.0.0, PTL-0050). https://protologue.com/t/llm-as-a-judge/"
}