{
  "id": "unfaithful-chain-of-thought",
  "code": "PTL-0101",
  "term": "Unfaithful Chain-of-Thought",
  "aliases": [
    "CoT faithfulness",
    "post-hoc rationalization"
  ],
  "category": "failure-modes",
  "definition": "Unfaithful chain-of-thought is stated reasoning that does not reflect the factors that actually determined the model's answer, so the explanation can be plausible yet misleading.",
  "description": "Turpin et al. showed that biasing features, such as always putting the correct answer in position A of few-shot examples, swayed answers while the stated reasoning never mentioned them. Lanham et al. measured how much answers actually depend on the stated reasoning, with results varying by task and model size.",
  "example": null,
  "broader": [],
  "narrower": [],
  "related": [
    "chain-of-thought",
    "reasoning-model"
  ],
  "introduced": null,
  "sources": [
    {
      "title": "Language Models Don't Always Say What They Think: Unfaithful Explanations in Chain-of-Thought Prompting",
      "authors": "Turpin et al.",
      "year": 2023,
      "url": "https://arxiv.org/abs/2305.04388"
    },
    {
      "title": "Measuring Faithfulness in Chain-of-Thought Reasoning",
      "authors": "Lanham et al.",
      "year": 2023,
      "url": "https://arxiv.org/abs/2307.13702"
    }
  ],
  "url": "https://protologue.com/t/unfaithful-chain-of-thought/",
  "citation": "Protologue. (2026). Unfaithful Chain-of-Thought. In Protologue: A Taxonomy of Prompting and LLM Techniques (v1.0.0, PTL-0101). https://protologue.com/t/unfaithful-chain-of-thought/"
}