{
  "id": "constitutional-ai",
  "code": "PTL-0095",
  "term": "Constitutional AI",
  "aliases": [
    "CAI",
    "RLAIF",
    "reinforcement learning from AI feedback"
  ],
  "category": "security",
  "definition": "Constitutional AI is a training method in which a model critiques and revises its own outputs according to a written set of principles, and AI-generated preference judgments replace most human labels for harmlessness.",
  "description": "Bai et al. used a supervised self-critique phase followed by reinforcement learning from AI feedback. The approach made the principles governing model behavior explicit and editable.",
  "example": null,
  "broader": [],
  "narrower": [],
  "related": [
    "rlhf",
    "self-refine",
    "jailbreak"
  ],
  "introduced": 2022,
  "sources": [
    {
      "title": "Constitutional AI: Harmlessness from AI Feedback",
      "authors": "Bai et al.",
      "year": 2022,
      "url": "https://arxiv.org/abs/2212.08073"
    }
  ],
  "url": "https://protologue.com/t/constitutional-ai/",
  "citation": "Protologue. (2026). Constitutional AI. In Protologue: A Taxonomy of Prompting and LLM Techniques (v1.0.0, PTL-0095). https://protologue.com/t/constitutional-ai/"
}