{
  "id": "prompt-compression",
  "code": "PTL-0080",
  "term": "Prompt Compression",
  "aliases": [
    "LLMLingua",
    "context compression"
  ],
  "category": "optimization",
  "definition": "Prompt compression shortens a prompt by removing tokens that contribute little information, typically scored by a smaller language model, to cut cost and latency while preserving task performance.",
  "description": "LLMLingua used a small model's perplexity to drop low-information tokens and reported high compression ratios with limited performance loss.",
  "example": null,
  "broader": [],
  "narrower": [],
  "related": [
    "context-engineering",
    "prompt-caching"
  ],
  "introduced": 2023,
  "sources": [
    {
      "title": "LLMLingua: Compressing Prompts for Accelerated Inference of Large Language Models",
      "authors": "Jiang et al.",
      "year": 2023,
      "url": "https://arxiv.org/abs/2310.05736"
    }
  ],
  "url": "https://protologue.com/t/prompt-compression/",
  "citation": "Protologue. (2026). Prompt Compression. In Protologue: A Taxonomy of Prompting and LLM Techniques (v1.0.0, PTL-0080). https://protologue.com/t/prompt-compression/"
}