diff --git a/prompt_compression/README.md b/prompt_compression/README.md new file mode 100644 index 00000000000..ee156a12fc4 --- /dev/null +++ b/prompt_compression/README.md @@ -0,0 +1,27 @@ +# Prompt Compression Plugins + +Microservices that plug into the LiteLLM gateway as `pre_call` guardrails to compress incoming prompts before they reach the LLM provider. Each runs independently and speaks the [LiteLLM generic guardrail API](https://docs.litellm.ai/docs/adding_provider/generic_guardrail_api). + +## How it works + +LiteLLM sends the full message array to the microservice before forwarding the request to the LLM provider. The microservice compresses it and returns the compressed messages. LiteLLM then sends the smaller payload to the provider — reducing cost and latency transparently to the caller. + +## Available plugins + +| Plugin | Compression approach | +|--------|----------------------| +| [headroom](./headroom/) | ML-based; deduplicates JSON arrays, compresses tool outputs and prose via a trained model | + +## Adding a plugin to your LiteLLM config + +```yaml +guardrails: + - guardrail_name: headroom-compression + litellm_params: + guardrail: generic_guardrail_api + mode: pre_call + api_base: http://localhost:8100 + # api_key: your-secret-key +``` + +The `mode: pre_call` ensures compression runs before the request reaches the provider. The plugin returns `GUARDRAIL_INTERVENED` with compressed `structured_messages` when savings are found, or `NONE` to pass through unchanged. diff --git a/prompt_compression/headroom/.env.example b/prompt_compression/headroom/.env.example new file mode 100644 index 00000000000..b7c35da2133 --- /dev/null +++ b/prompt_compression/headroom/.env.example @@ -0,0 +1,3 @@ +OPENAI_API_KEY=sk-... +HEADROOM_DEFAULT_MODEL=gpt-4o-mini +GUARDRAIL_API_KEY= diff --git a/prompt_compression/headroom/Dockerfile b/prompt_compression/headroom/Dockerfile new file mode 100644 index 00000000000..0aa20e4c2cc --- /dev/null +++ b/prompt_compression/headroom/Dockerfile @@ -0,0 +1,15 @@ +FROM python:3.12-slim + +WORKDIR /app + +COPY pyproject.toml . +RUN pip install --no-cache-dir ".[all]" 2>/dev/null || pip install --no-cache-dir \ + "fastapi>=0.115" \ + "uvicorn[standard]>=0.30" \ + "headroom-ai[all]>=0.1" + +COPY main.py . + +EXPOSE 8100 + +CMD ["uvicorn", "main:app", "--host", "0.0.0.0", "--port", "8100"] diff --git a/prompt_compression/headroom/README.md b/prompt_compression/headroom/README.md new file mode 100644 index 00000000000..e4a13c1274b --- /dev/null +++ b/prompt_compression/headroom/README.md @@ -0,0 +1,51 @@ +# Headroom prompt compression guardrail + +Wraps [headroom-ai](https://github.com/chopratejas/headroom) as a LiteLLM `pre_call` guardrail. Compresses tool outputs, JSON arrays, and prose in the message history before the request reaches the LLM provider. + +Headroom's `smart_crusher` deduplicates repeated JSON rows and converts verbose object arrays to a compact schema+CSV format. `kompress` handles prose and log compression. Typical savings: 40-90% on agentic workloads with large tool outputs. + +## Running locally + +```bash +cp .env.example .env # set OPENAI_API_KEY (headroom uses it internally) +docker compose up +``` + +Service listens on `http://localhost:8100`. + +## Environment variables + +| Variable | Required | Default | Description | +|----------|----------|---------|-------------| +| `OPENAI_API_KEY` | Yes | — | API key headroom uses for compression | +| `HEADROOM_DEFAULT_MODEL` | No | `gpt-4o-mini` | Model used for token budget calculations | +| `GUARDRAIL_API_KEY` | No | — | If set, requires `x-api-key` header on requests | + +## LiteLLM config + +```yaml +model_list: + - model_name: gpt-4o + litellm_params: + model: openai/gpt-4o + +guardrails: + - guardrail_name: headroom-compression + litellm_params: + guardrail: generic_guardrail_api + mode: pre_call + api_base: http://localhost:8100 + # api_key: your-secret-key +``` + +## What gets compressed + +- Tool results containing JSON (deduplication + schema compression) +- Tool results containing logs and prose +- User messages (opt-in via `compress_user_messages=True`, already enabled) + +Conversational turns under 250 tokens are left unchanged. Error outputs are protected from compression. + +## Deploying to Render + +The included `render.yaml` deploys as a Docker web service. Set `OPENAI_API_KEY` as a secret environment variable in the Render dashboard after the first deploy. diff --git a/prompt_compression/headroom/docker-compose.yml b/prompt_compression/headroom/docker-compose.yml new file mode 100644 index 00000000000..3ed61c9534a --- /dev/null +++ b/prompt_compression/headroom/docker-compose.yml @@ -0,0 +1,13 @@ +services: + headroom-guardrail: + build: . + ports: + - "8100:8100" + environment: + # LLM provider key headroom uses internally to compress + - OPENAI_API_KEY=${OPENAI_API_KEY} + # Optional: lock down the guardrail endpoint + - GUARDRAIL_API_KEY=${GUARDRAIL_API_KEY:-} + # Model headroom uses for compression (falls back to gpt-4o-mini) + - HEADROOM_DEFAULT_MODEL=${HEADROOM_DEFAULT_MODEL:-gpt-4o-mini} + restart: unless-stopped diff --git a/prompt_compression/headroom/litellm_config_example.yaml b/prompt_compression/headroom/litellm_config_example.yaml new file mode 100644 index 00000000000..c5b87eef9eb --- /dev/null +++ b/prompt_compression/headroom/litellm_config_example.yaml @@ -0,0 +1,12 @@ +model_list: + - model_name: gpt-4o + litellm_params: + model: openai/gpt-4o + +guardrails: + - guardrail_name: headroom-compression + litellm_params: + guardrail: generic_guardrail_api + mode: pre_call + api_base: http://localhost:8100 + # api_key: your-secret-key # matches GUARDRAIL_API_KEY in the service diff --git a/prompt_compression/headroom/main.py b/prompt_compression/headroom/main.py new file mode 100644 index 00000000000..e2e694cbf50 --- /dev/null +++ b/prompt_compression/headroom/main.py @@ -0,0 +1,96 @@ +from __future__ import annotations + +import logging +import os +from typing import Annotated, Any, Optional + +from fastapi import FastAPI, Header, HTTPException, status +from headroom import compress +from pydantic import BaseModel, Field + +logger = logging.getLogger(__name__) + +app = FastAPI(title="Headroom Guardrail", version="0.1.0") + +_API_KEY = os.environ.get("GUARDRAIL_API_KEY") + + +class GuardrailRequest(BaseModel): + input_type: str + texts: Optional[list[str]] = None + images: Optional[list[str]] = None + structured_messages: Optional[list[dict[str, Any]]] = None + tools: Optional[list[dict[str, Any]]] = None + tool_calls: Optional[list[dict[str, Any]]] = None + model: Optional[str] = None + litellm_call_id: Optional[str] = None + litellm_trace_id: Optional[str] = None + request_data: Optional[dict[str, Any]] = None + additional_provider_specific_params: Optional[dict[str, Any]] = Field( + default=None + ) + + +class GuardrailResponse(BaseModel): + action: str + texts: Optional[list[str]] = None + images: Optional[list[str]] = None + structured_messages: Optional[list[dict[str, Any]]] = None + blocked_reason: Optional[str] = None + + +def _resolve_model(request: GuardrailRequest) -> str: + if request.model: + return request.model + return os.environ.get("HEADROOM_DEFAULT_MODEL", "gpt-4o-mini") + + +@app.get("/health") +def health() -> dict[str, str]: + return {"status": "ok"} + + +@app.post("/beta/litellm_basic_guardrail_api", response_model=GuardrailResponse) +async def guardrail( + request: GuardrailRequest, + x_api_key: Annotated[Optional[str], Header()] = None, +) -> GuardrailResponse: + if _API_KEY and x_api_key != _API_KEY: + raise HTTPException(status_code=status.HTTP_401_UNAUTHORIZED, detail="Unauthorized") + + if request.input_type != "request": + return GuardrailResponse(action="NONE") + + messages = request.structured_messages + if not messages: + return GuardrailResponse(action="NONE") + + model = _resolve_model(request) + + try: + result = compress(messages, model=model, compress_user_messages=True, protect_recent=0) + except Exception: + logger.exception( + "headroom compress failed (call_id=%s); passing through unchanged", + request.litellm_call_id, + ) + return GuardrailResponse(action="NONE") + + compressed_messages: list[dict[str, Any]] = result.messages # type: ignore[attr-defined] + + tokens_saved = getattr(result, "tokens_saved", 0) + logger.info( + "headroom compressed call_id=%s model=%s tokens_saved=%s compression_ratio=%s", + request.litellm_call_id, + model, + tokens_saved, + getattr(result, "compression_ratio", "?"), + ) + + if not tokens_saved: + return GuardrailResponse(action="NONE") + + return GuardrailResponse( + action="GUARDRAIL_INTERVENED", + structured_messages=compressed_messages, + ) diff --git a/prompt_compression/headroom/pyproject.toml b/prompt_compression/headroom/pyproject.toml new file mode 100644 index 00000000000..58ebf35cf62 --- /dev/null +++ b/prompt_compression/headroom/pyproject.toml @@ -0,0 +1,9 @@ +[project] +name = "headroom-guardrail" +version = "0.1.0" +requires-python = ">=3.11" +dependencies = [ + "fastapi>=0.115", + "uvicorn[standard]>=0.30", + "headroom-ai[all]>=0.1", +] diff --git a/prompt_compression/headroom/render.yaml b/prompt_compression/headroom/render.yaml new file mode 100644 index 00000000000..bb099239139 --- /dev/null +++ b/prompt_compression/headroom/render.yaml @@ -0,0 +1,12 @@ +services: + - type: web + name: headroom-guardrail + runtime: docker + plan: starter + dockerfilePath: ./Dockerfile + envVars: + - key: HEADROOM_DEFAULT_MODEL + value: gpt-4o-mini + - key: GUARDRAIL_API_KEY + generateValue: true + healthCheckPath: /health