{
  "id": "process-reward-model",
  "code": "PTL-0053",
  "term": "Process Reward Model",
  "aliases": [
    "PRM",
    "step-level verifier",
    "process supervision"
  ],
  "category": "verification",
  "definition": "A process reward model (PRM) scores each intermediate step of a model's reasoning, rather than only the final answer, and is used to select or train better reasoning chains.",
  "description": "Lightman et al. showed process supervision outperformed outcome supervision for selecting correct solutions to competition math problems, and released a large dataset of step-level human labels.",
  "example": null,
  "broader": [],
  "narrower": [],
  "related": [
    "best-of-n-sampling",
    "test-time-compute-scaling",
    "llm-as-a-judge"
  ],
  "introduced": 2023,
  "sources": [
    {
      "title": "Let's Verify Step by Step",
      "authors": "Lightman et al.",
      "year": 2023,
      "url": "https://arxiv.org/abs/2305.20050"
    }
  ],
  "url": "https://protologue.com/t/process-reward-model/",
  "citation": "Protologue. (2026). Process Reward Model. In Protologue: A Taxonomy of Prompting and LLM Techniques (v1.0.0, PTL-0053). https://protologue.com/t/process-reward-model/"
}