mirror of
https://github.com/BerriAI/litellm.git
synced 2026-09-29 01:42:19 +00:00
Introduces a mitmproxy-based recording HTTP/HTTPS sidecar that any CI
job can opt into to cache LLM-provider responses across runs. Unlike
the in-process VCR persister at tests/_vcr_redis_persister.py — which
can only intercept HTTP traffic from the same Python process where it
was loaded — this sidecar operates at the network layer, so it works
for any e2e job whose system-under-test runs in a Docker container
(every job under e2e_*, proxy_*, etc.).
Components
- tests/e2e_cassette_proxy/cache_key.py: pure-function cache-key
derivation. Hashes (method, scheme, host, path, sorted query,
allowlisted headers, canonical-JSON body); strips auth, tracing,
and SDK-metadata headers so equivalent requests collide regardless
of run-to-run noise.
- tests/e2e_cassette_proxy/redis_store.py: thin Redis wrapper that
stores one (request, response) pair per key as MessagePack
(JSON+base64 fallback). Caps per-key payload size, drops oversize
responses with a log line, and never blocks the request path on
Redis errors.
- tests/e2e_cassette_proxy/addon.py: mitmproxy addon that ties the
two together. Hosts on the passthrough list (localhost, the proxy
itself) are never cached; non-2xx upstream responses are not
persisted.
- tests/e2e_cassette_proxy/Dockerfile: pinned python:3.12-slim base +
pinned mitmproxy 11.0.2 + pinned redis-py + pinned msgpack.
- tests/e2e_cassette_proxy/trust_ca.sh: helper for SUT containers to
trust the proxy CA in every Python / curl / boto3 / node trust
store at once.
- tests/e2e_cassette_proxy/README.md: usage guide + opt-in checklist
for other e2e jobs.
CI integration
- New reusable command 'start_cassette_proxy' in .circleci/config.yml.
Builds the image, runs the sidecar wired to the project Redis, fetches
the proxy CA, and exports CASSETTE_PROXY_URL / CASSETTE_PROXY_CA into
$BASH_ENV for downstream steps.
- e2e_openai_endpoints is wired up as the canonical demo: two-line opt-in
pattern documented in the README.
- Job logs include a 'Cassette-proxy stats' step that dumps the
hit/miss/store summary via 'docker logs cassette-proxy | grep
[E2ECASS]'.
Tests
- 31 hermetic unit tests under tests/test_litellm/e2e_cassette_proxy:
- test_cache_key.py: 14 tests pinning equivalence-class behavior of
the key derivation (auth header, tracing header, JSON key order,
query order, host case all collapse; method/path/body/allowlisted
headers / query-param values do not).
- test_redis_store.py: 9 tests covering set/get round-trip, default
TTL, binary body round-trip, oversize-payload rejection, corrupt-
blob eviction, and graceful behavior when the Redis client raises.
- test_addon.py: 8 tests using a fake-mitmproxy flow to exercise
the addon end-to-end (passthrough miss, persist on 2xx, hit on
canonicalized-equivalent re-request, no-persist on 5xx, host
passthrough, replay-only 599-on-miss, record-only never serves
cache).
- 31/31 pass.
Co-authored-by: Mateo Wang <mateo-berri@users.noreply.github.com>
158 lines
4.7 KiB
Python
158 lines
4.7 KiB
Python
"""Cache key derivation for the e2e recording proxy.
|
|
|
|
The proxy is intentionally promiscuous about *what* it caches: any HTTPS
|
|
egress that flows through it is a candidate. The cache key has to be
|
|
stable across runs that send equivalent requests, while staying robust
|
|
to per-call noise (auth headers, tracing IDs, dates).
|
|
|
|
We hash a canonical tuple of:
|
|
|
|
- HTTP method
|
|
- scheme + host + port
|
|
- URL path
|
|
- query string (sorted)
|
|
- request body (raw bytes; JSON bodies are re-serialized in canonical
|
|
form so that semantically-equal payloads collide)
|
|
- a small allowlist of headers the upstream actually keys on (e.g.
|
|
``content-type``, ``accept``)
|
|
|
|
Anything not on that allowlist is dropped before hashing. This is the
|
|
same trade-off vcrpy makes when configuring ``filter_headers`` —
|
|
strict enough to dedupe equivalent requests, loose enough to ignore
|
|
the auth/tracing churn that varies run-to-run.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import hashlib
|
|
import json
|
|
from typing import Iterable, Mapping, Optional, Sequence, Tuple
|
|
from urllib.parse import parse_qsl, urlsplit
|
|
|
|
CACHE_KEY_PREFIX = "litellm:e2ecass:"
|
|
|
|
# Headers that materially change what an upstream returns.
|
|
DEFAULT_HEADER_ALLOWLIST: Tuple[str, ...] = (
|
|
"accept",
|
|
"accept-encoding",
|
|
"content-type",
|
|
"openai-beta",
|
|
"anthropic-beta",
|
|
"anthropic-version",
|
|
"x-stainless-lang",
|
|
)
|
|
|
|
# Headers we *never* want in the key (auth, tracing, per-request noise).
|
|
DEFAULT_HEADER_BLOCKLIST: Tuple[str, ...] = (
|
|
"authorization",
|
|
"x-api-key",
|
|
"anthropic-api-key",
|
|
"openai-api-key",
|
|
"azure-api-key",
|
|
"api-key",
|
|
"cookie",
|
|
"user-agent",
|
|
"x-amz-security-token",
|
|
"x-amz-date",
|
|
"x-amz-content-sha256",
|
|
"amz-sdk-invocation-id",
|
|
"amz-sdk-request",
|
|
"x-goog-api-key",
|
|
"x-goog-user-project",
|
|
"x-request-id",
|
|
"request-id",
|
|
"traceparent",
|
|
"tracestate",
|
|
"x-stainless-arch",
|
|
"x-stainless-os",
|
|
"x-stainless-runtime",
|
|
"x-stainless-runtime-version",
|
|
"x-stainless-package-version",
|
|
"host",
|
|
"content-length",
|
|
"connection",
|
|
)
|
|
|
|
|
|
def _canonical_body(body: bytes) -> bytes:
|
|
"""JSON bodies often differ only in key order or whitespace; collapse
|
|
those into a single canonical form so the cache hits across sessions.
|
|
Non-JSON bodies are passed through unchanged."""
|
|
if not body:
|
|
return b""
|
|
try:
|
|
decoded = json.loads(body)
|
|
except (ValueError, UnicodeDecodeError):
|
|
return body
|
|
return json.dumps(decoded, sort_keys=True, separators=(",", ":")).encode("utf-8")
|
|
|
|
|
|
def _canonical_query(raw_query: str) -> str:
|
|
if not raw_query:
|
|
return ""
|
|
pairs = sorted(parse_qsl(raw_query, keep_blank_values=True))
|
|
return "&".join(f"{k}={v}" for k, v in pairs)
|
|
|
|
|
|
def _normalized_headers(
|
|
headers: Mapping[str, str],
|
|
allowlist: Sequence[str],
|
|
blocklist: Sequence[str],
|
|
) -> Tuple[Tuple[str, str], ...]:
|
|
allow = {h.lower() for h in allowlist}
|
|
block = {h.lower() for h in blocklist}
|
|
out: list[Tuple[str, str]] = []
|
|
for key, value in headers.items():
|
|
lk = key.lower()
|
|
if lk in block:
|
|
continue
|
|
if allow and lk not in allow:
|
|
continue
|
|
out.append((lk, value))
|
|
out.sort()
|
|
return tuple(out)
|
|
|
|
|
|
def derive_cache_key(
|
|
method: str,
|
|
url: str,
|
|
body: bytes,
|
|
headers: Mapping[str, str],
|
|
*,
|
|
allowlist: Optional[Iterable[str]] = None,
|
|
blocklist: Optional[Iterable[str]] = None,
|
|
) -> str:
|
|
"""Hash the canonicalized (method, url, body, allowlisted-headers) tuple
|
|
into a stable Redis key. Equal inputs (modulo header / JSON noise) always
|
|
produce the same key.
|
|
"""
|
|
parts = urlsplit(url)
|
|
canonical_headers = _normalized_headers(
|
|
headers,
|
|
allowlist=(
|
|
tuple(allowlist) if allowlist is not None else DEFAULT_HEADER_ALLOWLIST
|
|
),
|
|
blocklist=(
|
|
tuple(blocklist) if blocklist is not None else DEFAULT_HEADER_BLOCKLIST
|
|
),
|
|
)
|
|
canonical_body = _canonical_body(body)
|
|
digest = hashlib.sha256()
|
|
digest.update(method.upper().encode("ascii"))
|
|
digest.update(b"\x1f")
|
|
digest.update(parts.scheme.lower().encode("ascii"))
|
|
digest.update(b"://")
|
|
digest.update(parts.netloc.lower().encode("ascii"))
|
|
digest.update(b"\x1f")
|
|
digest.update(parts.path.encode("utf-8"))
|
|
digest.update(b"\x1f")
|
|
digest.update(_canonical_query(parts.query).encode("utf-8"))
|
|
digest.update(b"\x1f")
|
|
for k, v in canonical_headers:
|
|
digest.update(k.encode("ascii"))
|
|
digest.update(b"=")
|
|
digest.update(v.encode("utf-8", errors="replace"))
|
|
digest.update(b"\x1e")
|
|
digest.update(b"\x1f")
|
|
digest.update(canonical_body)
|
|
return f"{CACHE_KEY_PREFIX}{digest.hexdigest()}"
|