{
  "id": "rlhf",
  "code": "PTL-0011",
  "term": "Reinforcement Learning from Human Feedback",
  "aliases": [
    "RLHF",
    "preference tuning"
  ],
  "category": "foundations",
  "definition": "Reinforcement learning from human feedback (RLHF) trains a language model to produce outputs people prefer, by learning a reward model from human comparisons and optimizing the model against it.",
  "description": "InstructGPT showed that RLHF made a much smaller model preferred over a far larger base model. RLHF shapes how models respond to prompts, including their helpfulness and refusals, and is linked to failure modes such as sycophancy. Direct preference optimization is a widely used simpler alternative.",
  "example": null,
  "broader": [],
  "narrower": [
    "direct-preference-optimization"
  ],
  "related": [
    "instruction-tuning",
    "constitutional-ai",
    "sycophancy"
  ],
  "introduced": 2022,
  "sources": [
    {
      "title": "Training language models to follow instructions with human feedback",
      "authors": "Ouyang et al.",
      "year": 2022,
      "url": "https://arxiv.org/abs/2203.02155"
    }
  ],
  "url": "https://protologue.com/t/rlhf/",
  "citation": "Protologue. (2026). Reinforcement Learning from Human Feedback. In Protologue: A Taxonomy of Prompting and LLM Techniques (v1.0.0, PTL-0011). https://protologue.com/t/rlhf/"
}