test(e2e): move the harness self-tests out of tests/e2e (#45172)

* test(e2e): move the harness self-tests out of tests/e2e

The nightly Buildkite run copies tests/e2e into the runner image and runs
bare pytest, so the 672 tests of the harness itself (fixture parsing, JUnit
properties, the stack lock, the load aggregators, the Claude Code driver)
counted as e2e tests on the status page even though none of them reaches a
proxy. They now live in tests/e2e_harness, mirroring the tests/e2e layout,
and run in the GitHub Actions lint job and the CircleCI
provider_replay_harness job instead

* fix(ci): point the providers replay controls at tests/e2e_harness

The providers integration job still selected the four replay-control
tests under tests/e2e/test_provider_edge.py, so pytest exited before
they ran. The raw-HTTP check's file walk also drops to one loop per
comprehension

* style(tests): mark the raw-HTTP check's bindings Final
This commit is contained in:
ryan-crabbe-berri 2026-10-07 15:20:02 -07:00 • committed by GitHub
parent 2aaa0b5d5c
commit 048a1500df
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
49 changed files with 183 additions and 144 deletions

View file

@ -3145,10 +3145,10 @@ jobs:
name: Test provider capture and replay harness
command: |
mkdir -p test-results/provider-replay-harness
uv run --no-sync pytest --tb=short -q --noconftest -o addopts= -o pythonpath=tests/e2e -p no:rerunfailures \
uv run --no-sync pytest --tb=short -q --noconftest -o addopts= -o "pythonpath=tests/e2e tests/e2e_harness" -p no:rerunfailures \
--junitxml=test-results/provider-replay-harness/junit.xml \
tests/e2e/test_provider_edge.py tests/e2e/test_fixture_bundle.py \
tests/e2e/test_fixture_canonical.py tests/e2e/test_fixture_mode.py \
tests/e2e_harness/test_provider_edge.py tests/e2e_harness/test_fixture_bundle.py \
tests/e2e_harness/test_fixture_canonical.py tests/e2e_harness/test_fixture_mode.py \
tests/code_coverage_tests/test_provider_replay_harness.py \
tests/code_coverage_tests/test_provider_cache.py
- store_test_results:

View file

@ -24,8 +24,8 @@ while IFS= read -r file || [ -n "$file" ]; do
has_mcp_dependencies=true ;;
esac
case "$file" in
tests/e2e/*/*.py) : ;;
tests/e2e/*.py | tests/code_coverage_tests/test_provider_cache.py | tests/code_coverage_tests/test_provider_replay_harness.py | tests/unit/test_circleci_path_filter.py | .circleci/* | pyproject.toml | uv.lock)
tests/e2e/*/*.py | tests/e2e_harness/*/*.py) : ;;
tests/e2e/*.py | tests/e2e_harness/*.py | tests/code_coverage_tests/test_provider_cache.py | tests/code_coverage_tests/test_provider_replay_harness.py | tests/unit/test_circleci_path_filter.py | .circleci/* | pyproject.toml | uv.lock)
has_provider_harness=true ;;
esac
case "$file" in

View file

@ -193,10 +193,10 @@ fi
if [ "$suite" = providers ]; then
INTEGRATION_RUN_ID="$integration_identity" .venv/bin/python -m pytest --tb=short --noconftest -o addopts= \
--strict-markers --strict-config -p no:pytest-retry -p no:rerunfailures --timeout=30 \
tests/e2e/test_provider_edge.py::TestReplayMode::test_content_drift_returns_the_miss_status_naming_both_keys \
tests/e2e/test_provider_edge.py::TestReplayMode::test_exhausted_key_returns_the_miss_status \
tests/e2e/test_provider_edge.py::TestReplayLeftover::test_partially_consumed_recording_names_the_leftover \
tests/e2e/test_provider_edge.py::TestStreamingFidelity::test_replay_of_a_stream_makes_no_provider_connection \
tests/e2e_harness/test_provider_edge.py::TestReplayMode::test_content_drift_returns_the_miss_status_naming_both_keys \
tests/e2e_harness/test_provider_edge.py::TestReplayMode::test_exhausted_key_returns_the_miss_status \
tests/e2e_harness/test_provider_edge.py::TestReplayLeftover::test_partially_consumed_recording_names_the_leftover \
tests/e2e_harness/test_provider_edge.py::TestStreamingFidelity::test_replay_of_a_stream_makes_no_provider_connection \
--junitxml="$results/replay-controls.xml"
fi

View file

@ -174,23 +174,25 @@ jobs:
- name: Check tests/e2e basedpyright (zero errors)
if: steps.changes.outputs.decision != 'skip'
run: |
if git diff --name-only --diff-filter=ACMRD "$GATE_BASE_SHA" HEAD -- ':(glob)tests/e2e/**/*.py' | grep -q .; then
uv run --no-sync basedpyright tests/e2e
if git diff --name-only --diff-filter=ACMRD "$GATE_BASE_SHA" HEAD -- ':(glob)tests/e2e/**/*.py' ':(glob)tests/e2e_harness/**/*.py' pyrightconfig.json | grep -q .; then
uv run --no-sync basedpyright tests/e2e tests/e2e_harness
else
echo "No changed tests/e2e Python files; skipping."
fi
- name: Run the claude_code harness unit tests
- name: Run the e2e harness tests
if: steps.changes.outputs.decision != 'skip'
env:
LITELLM_MASTER_KEY: sk-e2e-harness-tests-reach-no-proxy
run: |
if ! git diff --name-only --diff-filter=ACMRD "$GATE_BASE_SHA" HEAD -- ':(glob)tests/e2e/claude_code/**/*.py' ':(glob)tests/e2e/*.py' tests/e2e/claude_code/cron_vm/install_claude_code.sh pyproject.toml uv.lock .github/workflows/test-linting.yml | grep -q .; then
echo "No changed claude_code harness files; skipping."
if ! git diff --name-only --diff-filter=ACMRD "$GATE_BASE_SHA" HEAD -- tests/e2e tests/e2e_harness ':(exclude)tests/e2e/ui' pyproject.toml uv.lock .github/workflows/test-linting.yml | grep -q .; then
echo "No changed e2e harness files; skipping."
exit 0
fi
retry() { "$@" || { sleep 15; "$@"; } || { sleep 30; "$@"; }; }
CLAUDE_VERSION="$(retry uv run --no-sync python tests/e2e/claude_code/pr_gate_version_resolver.py)"
tests/e2e/claude_code/cron_vm/install_claude_code.sh "$CLAUDE_VERSION" "$RUNNER_TEMP/claude-cli"
PATH="$RUNNER_TEMP/claude-cli:$PATH" uv run --no-sync pytest -q --noconftest -o addopts= -o pythonpath=tests/e2e -p no:rerunfailures tests/e2e/claude_code/_*_unit_tests
PATH="$RUNNER_TEMP/claude-cli:$PATH" uv run --no-sync pytest -q tests/e2e_harness
- name: Check for circular imports
if: steps.changes.outputs.decision != 'skip'

View file

@ -31,7 +31,7 @@ help:
@echo " make lint - Run all linting (Ruff, basedpyright, format check, circular imports, import safety)"
@echo " make lint-ruff - Run Ruff linting only"
@echo " make lint-basedpyright - Run basedpyright strict, gated by per-rule error counts"
@echo " make lint-e2e-basedpyright - Run basedpyright over tests/e2e (zero errors allowed)"
@echo " make lint-e2e-basedpyright - Run basedpyright over tests/e2e and tests/e2e_harness (zero errors allowed)"
@echo " make lint-basedpyright-budget-update - Ratchet basedpyright limits down by what this branch fixed"
@echo " make lint-format - Check ruff format formatting (matches CI)"
@echo " make lint-ruff-budget - Gate the codebase total of each strict ruff rule against its limit"
@ -211,7 +211,7 @@ lint-basedpyright: $(LINT_DEP_INSTALL) $(LINT_DEP_BASE)
$(UV_RUN) python scripts/type_check_gate.py --base "$(BASE_REF)"
lint-e2e-basedpyright: $(LINT_E2E_DEP_INSTALL)
$(UV_RUN) basedpyright tests/e2e
$(UV_RUN) basedpyright tests/e2e tests/e2e_harness
# Type-discipline budget (mutable collections / casts / type guards / kwargs /
# unexplained suppressions), the test-linting.yml step `make lint` used to omit.

View file

@ -1,7 +1,13 @@
{
"include": ["litellm"],
"ignore": [],
"exclude": ["**/node_modules", "**/__pycache__", "tests/e2e/claude_code", "tests/e2e/ui", "litellm/types/utils.py", "litellm/proxy/_types.py"],
"exclude": ["**/node_modules", "**/__pycache__", "tests/e2e/claude_code", "tests/e2e_harness/claude_code", "tests/e2e/ui", "litellm/types/utils.py", "litellm/proxy/_types.py"],
"executionEnvironments": [
{
"root": "tests/e2e_harness",
"extraPaths": ["tests/e2e", "tests/e2e/batches", "tests/e2e/guardrails", "tests/e2e/load", "tests/e2e/logging"]
}
],
"pythonVersion": "3.12",
"typeCheckingMode": "strict",
"enableTypeIgnoreComments": false,

View file

@ -10,7 +10,8 @@
# with origin's current default branch, untracked files included
# The per-area checks:
# - litellm/ Python -> `make lint` (test-linting.yml's lint job)
# - tests/e2e Python -> `make lint-e2e-basedpyright` (test-linting.yml's e2e type-check step)
# - tests/e2e and tests/e2e_harness Python
# -> `make lint-e2e-basedpyright` (test-linting.yml's e2e type-check step)
# + raw HTTP client ban (test-code-quality.yml's check_e2e_no_raw_requests)
# - tests/ Python, ruff-tests.toml, test-quality-budget.json, scripts/check_test_quality.py,
# scripts/test_quality_gate.py
@ -95,7 +96,7 @@ existing_files() {
}
litellm_py_pattern='^litellm/.*\.py$'
e2e_py_pattern='^tests/e2e/.*\.py$'
e2e_py_pattern='^tests/e2e(_harness)?/.*\.py$'
test_tree_pattern='^(tests/.*\.py|ruff-tests\.toml|test-quality-budget\.json|scripts/(check_test_quality|test_quality_gate)\.py)$'
spec_pattern='^(litellm/(proxy|types)/.*|ui/litellm-dashboard/(scripts/gen-api-types\.mjs|package\.json|package-lock\.json|src/lib/http/schema\.d\.ts))$'
ui_prettier_pattern='^ui/litellm-dashboard/.*\.(js|jsx|ts|tsx|mjs|cjs|json|css|scss|md|mdx|yml|yaml|html)$'

View file

@ -1,7 +1,8 @@
# Tests
Nothing on the other side of the call: `tests/unit`. A proxy we start with an upstream we script:
`tests/integration`. Someone else's service with real credentials: `tests/e2e`. Two fit, split it
`tests/integration`. Someone else's service with real credentials: `tests/e2e`. Tests of that harness
itself, no proxy at all: `tests/e2e_harness`. Two fit, split it
## What good looks like

View file

@ -1,27 +1,31 @@
"""tests/e2e routes every HTTP call through the typed transport (e2e_http.py), so
raw HTTP client imports (requests, urllib.request, httpx, aiohttp, http.client) are
banned in suite code. Importing requests' exception types for catching is fine
anywhere; a small allowlist grandfathers the files that legitimately make raw calls
(the transport itself, the root conftest liveness probe, the claude_code version
resolver's constant registry URL fetch, and the mcp OAuth client, whose httpx
client is the object the official mcp SDK's streamable_http_client requires and so
cannot go through the sync requests transport). Referenced by tests/e2e/AGENTS.md."""
banned in suite code, and tests/e2e_harness, which tests that harness, is held to the
same ban. Importing requests' exception types for catching is fine anywhere; a small
allowlist grandfathers the files that legitimately make raw calls (the transport
itself, the root conftest liveness probe, the claude_code version resolver's constant
registry URL fetch, and the mcp OAuth client, whose httpx client is the object the
official mcp SDK's streamable_http_client requires and so cannot go through the sync
requests transport). Referenced by tests/e2e/AGENTS.md."""
from __future__ import annotations
import ast
import sys
from itertools import chain
from pathlib import Path
from typing import Final
E2E_DIR = Path(__file__).resolve().parents[1] / "e2e"
TESTS_DIR = Path(__file__).resolve().parents[1]
SCANNED_DIRS = ("e2e", "e2e_harness")
BANNED_MODULES = ("requests", "urllib.request", "http.client", "httpx", "aiohttp")
ALLOWED_RAW_CLIENT_FILES = {
"e2e_http.py": ("requests",),
"conftest.py": ("requests",),
"claude_code/pr_gate_version_resolver.py": ("urllib.request",),
"mcp/oauth_chat_client.py": ("httpx",),
"e2e/e2e_http.py": ("requests",),
"e2e/conftest.py": ("requests",),
"e2e/claude_code/pr_gate_version_resolver.py": ("urllib.request",),
"e2e/mcp/oauth_chat_client.py": ("httpx",),
}
EXCEPTION_ONLY_NAMES = frozenset({"RequestException", "ConnectionError", "Timeout", "HTTPError"})
@ -51,27 +55,28 @@ def _banned_imports(tree: ast.Module) -> tuple[tuple[str, int], ...]:
def _violations_in(path: Path) -> tuple[str, ...]:
relative = path.relative_to(E2E_DIR).as_posix()
relative = path.relative_to(TESTS_DIR).as_posix()
allowed = ALLOWED_RAW_CLIENT_FILES.get(relative, ())
tree = ast.parse(path.read_text(encoding="utf-8"), filename=str(path))
return tuple(
f"tests/e2e/{relative}:{lineno}: raw HTTP client import '{module}'"
f"tests/{relative}:{lineno}: raw HTTP client import '{module}'"
for module, lineno in _banned_imports(tree)
if module not in allowed
)
def _scanned_files() -> tuple[Path, ...]:
trees: Final = (sorted((TESTS_DIR / scanned).rglob("*.py")) for scanned in SCANNED_DIRS)
return tuple(chain.from_iterable(trees))
def main() -> int:
violations = tuple(
violation
for path in sorted(E2E_DIR.rglob("*.py"))
for violation in _violations_in(path)
)
violations: Final = tuple(chain.from_iterable(_violations_in(path) for path in _scanned_files()))
for violation in violations:
print(violation)
if violations:
print(
f"\n{len(violations)} raw HTTP client import(s) in tests/e2e. "
f"\n{len(violations)} raw HTTP client import(s) in tests/e2e or tests/e2e_harness. "
"Route the call through tests/e2e/e2e_http.py (get_external for absolute "
"third-party URLs) so it gets the typed Result handling."
)

View file

@ -256,7 +256,9 @@ assert replay_leftover_error(mode_raw="replay", bundle_dir=Path(sys.argv[1]), te
],
env={
**os.environ,
"PYTHONPATH": str(Path(__file__).resolve().parents[1] / "e2e"),
"PYTHONPATH": os.pathsep.join(
str(Path(__file__).resolve().parents[1] / tree) for tree in ("e2e", "e2e_harness")
),
"E2E_REPLAY_MATCH_PROFILE": "stateless_v1",
},
capture_output=True,

View file

@ -44,11 +44,11 @@ Each subdirectory under `tests/e2e/` is one suite, scoped to an endpoint family
- `logging/` - logging-integration delivery (datadog and friends)
- `security/` - secret handling and log-leak protection
- `router/` - routing and reliability behavior (fallbacks, cooldowns) plus the memory tests (`test_reliability_memory_e2e.py`: every worker's RSS as read at collection time, before any test traffic, must sit under a fixed idle budget, the release-gate check for a DB-backed boot that idles near the pod limit the way v1.100.x did; and a few hundred failing requests with retries and fallbacks must not grow proxy RSS past a fixed budget nor store a request snapshot past a fixed size, the release-gate check for the v1.100.0 retry-breadcrumb leak)
- `load/` - performance-category tests, kept OUT of the main suite: throughput/load SLO tests are a different testing category from functional e2e (variance-driven, historically flaky) and live outside this suite until re-implemented as their own pipeline (LIT-5163); do not add a live load test that runs in the default collection. What lives here: the weekly session-anomaly test (`test_weekly_session_anomaly_e2e.py`, Claude Code-shaped multi-turn sessions against real providers with ceilings on error rate, cache read/write, turn time, and spend; marked `weekly` and deselected unless `E2E_WEEKLY_ANOMALY` is set, driven by `.github/workflows/weekly_load_anomaly.yml`), the Redis chaos test (`test_redis_chaos_e2e.py`, locust load against mock deployments split round robin over `/chat/completions` and `/v1/messages`, one endpoint per simulated user, with `CLIENT PAUSE ALL` on the proxy's Redis mid-run to simulate it being down outright, asserting zero failed requests on every endpoint, budgeting RSS and CPU-per-request as ratios against the same run's healthy phase, and holding p50/p90/p99 latency and log-bytes-per-request to flat ceilings (a ratio cannot bound those two: an open breaker skips Redis instead of waiting on it, so the chaos phase can measure cheaper than baseline while still being far slower than a user should see); needs a proxy booted from `gateway/redis_chaos_ci_config.yml` on the same host with `E2E_PROXY_PID` and `E2E_PROXY_LOG` set, marked `redis_chaos`, deselected unless `E2E_REDIS_CHAOS` is set and excluded from the per-PR selector like the rest of `load/`, driven by `.github/workflows/test-e2e-redis-chaos.yml` and by the Buildkite `e2e-redis-chaos` step in project-releaser, which runs the proxy, Postgres and Valkey co-located with pytest in one pod and sets the opt-in), and markerless harness unit tests for the locust, process-usage, and session-anomaly aggregation logic
- `load/` - performance-category tests, kept OUT of the main suite: throughput/load SLO tests are a different testing category from functional e2e (variance-driven, historically flaky) and live outside this suite until re-implemented as their own pipeline (LIT-5163); do not add a live load test that runs in the default collection. What lives here: the weekly session-anomaly test (`test_weekly_session_anomaly_e2e.py`, Claude Code-shaped multi-turn sessions against real providers with ceilings on error rate, cache read/write, turn time, and spend; marked `weekly` and deselected unless `E2E_WEEKLY_ANOMALY` is set, driven by `.github/workflows/weekly_load_anomaly.yml`), the Redis chaos test (`test_redis_chaos_e2e.py`, locust load against mock deployments split round robin over `/chat/completions` and `/v1/messages`, one endpoint per simulated user, with `CLIENT PAUSE ALL` on the proxy's Redis mid-run to simulate it being down outright, asserting zero failed requests on every endpoint, budgeting RSS and CPU-per-request as ratios against the same run's healthy phase, and holding p50/p90/p99 latency and log-bytes-per-request to flat ceilings (a ratio cannot bound those two: an open breaker skips Redis instead of waiting on it, so the chaos phase can measure cheaper than baseline while still being far slower than a user should see); needs a proxy booted from `gateway/redis_chaos_ci_config.yml` on the same host with `E2E_PROXY_PID` and `E2E_PROXY_LOG` set, marked `redis_chaos`, deselected unless `E2E_REDIS_CHAOS` is set and excluded from the per-PR selector like the rest of `load/`, driven by `.github/workflows/test-e2e-redis-chaos.yml` and by the Buildkite `e2e-redis-chaos` step in project-releaser, which runs the proxy, Postgres and Valkey co-located with pytest in one pod and sets the opt-in). Its aggregation logic (locust, process usage, session anomaly) is covered by `tests/e2e_harness/load/`
- `other/` - the holding-pen suite for the `other.*` registry cluster with no home of its own yet: the master-key auth gate, JWT auth (access tokens issued by a real Keycloak realm, `idp.py` plus `idp_realm.json`, whose JWKS the proxy's `JWT_PUBLIC_KEY_URL` points at; see CONTRIBUTING.md for the start command and config block), and the process-lifecycle health probes (liveness, public readiness, authenticated readiness diagnostics). Promote a cluster out once it is large/stable enough for its own suite
- `secret_manager/` - the gateway's `key_management_system` against a real secret manager: deployment keys resolved from it (`os.environ/<name>` where the name exists only in the manager) and virtual keys written to and deleted from it. The tests are backend-agnostic and each backend is its own lane, because the setting is global to the proxy: `E2E_SECRET_MANAGER=<system>` opts in and picks the backend from `secret_backends.BACKENDS`, the proxy is booted from `gateway/secret_manager_<system>_ci_config.yml` against the live manager, and the tests reach that manager through the backend's `SecretStore` (`secret_store_<system>.py`). A test needing something not every backend does carries `requires_capability(...)` and is deselected on lanes that lack it. `secret_manager/backend.sh up <system>` runs a backend in Docker and writes the proxy's and the tests' env. Marked `secret_manager`, deselected unless `E2E_SECRET_MANAGER` is set, and kept out of the per-PR selector. Backends today: `hashicorp_vault` and `cyberark` (CyberArk Conjur, which cannot delete, so the delete test is Vault-only)
- `gateway/` - proxy configuration only (`litellm-config.yml`); no tests
- `claude_code/` - the Claude Code compatibility matrix: drives the real `claude` CLI (and HTTP probes) against a proxy for each feature x provider cell, reporting tagged-union outcomes via the `compat_result` fixture; ships its own driver/builder/publisher plus `_*_unit_tests/` trees. The HTTP probes ride the shared transport (`ProxyClient.count_tokens` / `ProxyClient.messages`); the CLI-driving path stays bespoke
- `claude_code/` - the Claude Code compatibility matrix: drives the real `claude` CLI (and HTTP probes) against a proxy for each feature x provider cell, reporting tagged-union outcomes via the `compat_result` fixture; ships its own driver/builder/publisher, covered by `tests/e2e_harness/claude_code/`. The HTTP probes ride the shared transport (`ProxyClient.count_tokens` / `ProxyClient.messages`); the CLI-driving path stays bespoke
- `ui/` - the Admin UI browser suite: Playwright in TypeScript, driving the dashboard served by a live proxy on port 4000 (seeded postgres + mock LLM upstream; see its `run_e2e.sh`). It is a self-contained npm package with its own lockfile and does not use the Python harness, pytest markers, or the shared transport; the Python rules in this file (typed models, `Result` unions, basedpyright zero-error gate) do not apply inside it. Its only Python file, `fixtures/mock_llm_server/server.py`, is excluded from the e2e basedpyright gate via the root `pyrightconfig.json`
## MCP suite: real Datadog only
@ -98,7 +98,7 @@ Each suite provides its own `client` fixture (see `llm_translation/passthrough_c
Request and response bodies are typed pydantic models in `models.py`; only the fields a test reads are modelled, and nothing passes raw dicts. Outcomes come back as a `Result[R]` tagged union (`Success`, `NetworkError`, `UnauthorizedError`, `RateLimitedError`, `ValidationError`, `UnknownApiError`). Handle them with `match`, or call `unwrap(...)` when a non-success should fail the test. The harness hard-fails and never skips: a test marked `e2e` fails when no proxy answers its liveness probe, and once a request reaches the proxy any wrong behavior is likewise a hard failure, so a missing proxy turns the run red instead of being mistaken for a pass
Mark live tests with `@pytest.mark.e2e` (on the class or the module). Coverage of the harness itself carries no marker and runs whether or not a proxy is up. Add `@pytest.mark.quiet_stack` to a test that measures the proxy itself (RSS, latency): the shared stack lock in `stack_lock.py` then runs it while no other test on the host is hitting the stack, marked or not, so the reading depends only on the test's own traffic. Use `scoped_key` for a fresh all-models key that auto-deletes, `resources` when you need to create and tear down more than a key, and `unique_marker()` from `e2e_config` to keep prompts, tags, and customer ids from colliding across concurrent runs and the shared response cache
Mark live tests with `@pytest.mark.e2e` (on the class or the module). Coverage of the harness itself lives outside the suite in `tests/e2e_harness/` (see its `AGENTS.md`) and runs without a proxy, so nothing under `tests/e2e/` is a markerless test. Add `@pytest.mark.quiet_stack` to a test that measures the proxy itself (RSS, latency): the shared stack lock in `stack_lock.py` then runs it while no other test on the host is hitting the stack, marked or not, so the reading depends only on the test's own traffic. Use `scoped_key` for a fresh all-models key that auto-deletes, `resources` when you need to create and tear down more than a key, and `unique_marker()` from `e2e_config` to keep prompts, tags, and customer ids from colliding across concurrent runs and the shared response cache
## Record and replay fixtures
@ -150,7 +150,7 @@ def test_bare_key_blocks_over_its_own_budget(...) -> None: ...
`route` is the endpoint the test is checking: `TEAM_MANAGEMENT` for a `/team/update` test, `SPEND_REPORTING` for a `/spend/logs` test, `MESSAGES` for a test of spend on `/v1/messages`. A test whose chat call only triggers the behavior under test, like the budget block above, leaves it unset, since its steps already name the call
Every pytest test in the live suites declares a `Subject` with at least its `domain`. Only the markerless harness tests (the root-level `test_*.py` files, `coverage_registry/`, `batches/test_batch_cleanup.py`, `guardrails/test_guardrails_client.py`, `logging/test_datadog_reader.py`, `logging/test_span_selection.py` and claude_code's `_*_unit_tests/`) and the `load/` suite carry none, since they drive nothing. The fields themselves stay optional, since a test that makes no LLM call has no provider, model or mode to name. Every field is a closed enum, so a typo is a basedpyright error at the call site rather than a property that silently never appears. `providers`, `models` and `capabilities` are tuples even with one member, because one test node routinely drives several: the claude_code matrix runs haiku, sonnet and opus in a single body, and a spend test calls two providers on one key. Declare every provider and every model the test drives, fallbacks included. The three are independent sets with no positional pairing between them (one provider x three models is the common case), and each is deduped and sorted at declaration so the committed run artifacts diff cleanly. `models=("gpt-5.5")` is a str and not a tuple, so anything but a tuple raises a `TypeError` where the decorator runs and shows up as a collection error naming the file. `Subject` is serialized with `dataclasses.asdict`, so a new scalar field needs no serializer edit; empty fields emit no `<property>` at all. A declared model names the constant the test drives (`CHEAP_ANTHROPIC_MODEL`, the file's own `BACKEND`), never a copy of its value, so the property cannot claim one model while an env override runs another. `e2e_metadata` and its call sites never import litellm, only the stdlib, pytest and pydantic: `Provider` mirrors litellm's `LlmProviders` values instead of importing them, because tests/e2e is shipped to the runner image on its own and a `from litellm...` at module scope would make the litellm package a hard dependency of COLLECTING the suite. `TestProviderMirrorsLitellm` in `tests/code_coverage_tests/test_e2e_metadata.py` fails on drift wherever litellm is importable and skips where it is not, so adding a provider is one line in `e2e_metadata`
Every pytest test under `tests/e2e/` declares a `Subject` with at least its `domain`, except the `load/` suite, which is kept out of the default collection. The harness's own tests in `tests/e2e_harness/` carry none, since they drive nothing. The fields themselves stay optional, since a test that makes no LLM call has no provider, model or mode to name. Every field is a closed enum, so a typo is a basedpyright error at the call site rather than a property that silently never appears. `providers`, `models` and `capabilities` are tuples even with one member, because one test node routinely drives several: the claude_code matrix runs haiku, sonnet and opus in a single body, and a spend test calls two providers on one key. Declare every provider and every model the test drives, fallbacks included. The three are independent sets with no positional pairing between them (one provider x three models is the common case), and each is deduped and sorted at declaration so the committed run artifacts diff cleanly. `models=("gpt-5.5")` is a str and not a tuple, so anything but a tuple raises a `TypeError` where the decorator runs and shows up as a collection error naming the file. `Subject` is serialized with `dataclasses.asdict`, so a new scalar field needs no serializer edit; empty fields emit no `<property>` at all. A declared model names the constant the test drives (`CHEAP_ANTHROPIC_MODEL`, the file's own `BACKEND`), never a copy of its value, so the property cannot claim one model while an env override runs another. `e2e_metadata` and its call sites never import litellm, only the stdlib, pytest and pydantic: `Provider` mirrors litellm's `LlmProviders` values instead of importing them, because tests/e2e is shipped to the runner image on its own and a `from litellm...` at module scope would make the litellm package a hard dependency of COLLECTING the suite. `TestProviderMirrorsLitellm` in `tests/code_coverage_tests/test_e2e_metadata.py` fails on drift wherever litellm is importable and skips where it is not, so adding a provider is one line in `e2e_metadata`
Declared fields ride out as JUnit `<property>` entries behind the fixed prefix, the same way steps do: each scalar under its field name, and each plural value as a repeated property under its SINGULAR name (`provider`, `model`, `capability`). The results JSON downstream regroups them under the plural key, so `providers`, `models` and `capabilities` are arrays there, `[]` when empty
@ -313,7 +313,7 @@ other.<area>.<case>.<assertion>
```
## Hard Rules
- no unit tests of a product feature under `tests/e2e`, and no mock tests or monkeypatching of code anywhere in it: a product feature is proven end to end against a live proxy, never with a unit test. if a contributor asks you to write an end to end test, do NOT stage a unit test with it; if you find a product gap, call it out in the PR description. the harness's own plumbing is the one exception: the markerless tests in the root-level `test_*.py` files, `coverage_registry/test_collector.py`, `guardrails/test_guardrails_client.py`, the `claude_code/_*_unit_tests/` trees, and the `load/` aggregation tests carry no `e2e` marker, run without a proxy, and take their inputs as arguments or env vars (setting an env var through pytest's `monkeypatch` fixture is fine, patching a function, class, or module is not), and no coverage-registry or compat-matrix cell rests on them. judge a change inside one of them by that standard, not as a misplaced product test
- no unit tests of a product feature under `tests/e2e`, and no mock tests or monkeypatching of code anywhere in it: a product feature is proven end to end against a live proxy, never with a unit test. if a contributor asks you to write an end to end test, do NOT stage a unit test with it; if you find a product gap, call it out in the PR description. the harness's own plumbing is tested outside the suite, in `tests/e2e_harness/` (mirroring this folder's layout), because the Buildkite e2e run copies `tests/e2e/` into the runner image and runs every test in it, so a harness test in here would count as a product test in the nightly numbers. those tests run without a proxy and take their inputs as arguments or env vars (setting an env var through pytest's `monkeypatch` fixture is fine, patching a function, class, or module is not), and no coverage-registry or compat-matrix cell rests on them. judge a change inside one of them by that standard, not as a misplaced product test
- use model management endpoints to create new models for a test. this could be in a conftest / inline for each test. ask the user what they want.

View file

@ -229,13 +229,13 @@ Each suite provides its own `client` fixture (see `llm_translation/passthrough_c
Request and response bodies are typed pydantic models in `models.py`; only the fields a test reads are modelled, and nothing passes raw dicts. Outcomes come back as a `Result[R]` tagged union (`Success`, `NetworkError`, `UnauthorizedError`, `RateLimitedError`, `ValidationError`, `UnknownApiError`). Handle them with `match`, or call `unwrap(...)` when a non-success should fail the test. The harness hard-fails and never skips: a test marked `e2e` fails when no proxy answers its liveness probe, and once a request reaches the proxy any wrong behavior is likewise a hard failure, so a missing proxy turns the run red instead of being mistaken for a pass
Mark live tests with `@pytest.mark.e2e` (on the class or the module). Pure coverage of the harness itself carries no marker and runs regardless. A test that needs proxy configuration the default stack does not carry goes behind an opt-in marker (`managed_files`, `prompt_caching_stack`, `weekly`), each deselected unless its env var is set; `OPT_IN_MARKERS` in `conftest.py` maps marker to env var, and the coverage collector counts such a cell only where the env var is set. Use `scoped_key` for a fresh all-models key that auto-deletes, `resources` when you need to create and tear down more than a key, and `unique_marker()` from `e2e_config` to keep prompts, tags, and customer ids from colliding across concurrent runs and the shared response cache
Mark live tests with `@pytest.mark.e2e` (on the class or the module). Pure coverage of the harness itself lives in `tests/e2e_harness/` and runs without a proxy (`LITELLM_MASTER_KEY=sk-harness uv run pytest tests/e2e_harness`). A test that needs proxy configuration the default stack does not carry goes behind an opt-in marker (`managed_files`, `prompt_caching_stack`, `weekly`), each deselected unless its env var is set; `OPT_IN_MARKERS` in `conftest.py` maps marker to env var, and the coverage collector counts such a cell only where the env var is set. Use `scoped_key` for a fresh all-models key that auto-deletes, `resources` when you need to create and tear down more than a key, and `unique_marker()` from `e2e_config` to keep prompts, tags, and customer ids from colliding across concurrent runs and the shared response cache
## Pre-commit steps
Before you push
1. Run `make lint-e2e-basedpyright` (or `make check` with your changes staged); the harness is fully typed and the gate allows zero basedpyright errors, enforced in CI on any PR touching `tests/e2e/**/*.py`
1. Run `make lint-e2e-basedpyright` (or `make check` with your changes staged); the harness is fully typed and the gate allows zero basedpyright errors, enforced in CI on any PR touching `tests/e2e/**/*.py` or `tests/e2e_harness/**/*.py`
2. Add the models your test needs to the config your local proxy loads
@ -260,7 +260,7 @@ The semantic header set is `content-type`, `accept`, `anthropic-version`, `anthr
Excluded transport and telemetry headers are `host`, `content-length`, `connection`, `accept-encoding`, `user-agent`, `traceparent`, `tracestate`, `x-request-id`, `x-client-request-id` and `x-stainless-*`. Inbound transfer-encoding is unsupported; send JSON with content-length framing. The destination represents host identity and the relay carries original body bytes. Replay does not verify credentials, SDK timeout/retry behavior, transport performance, model availability or stateful remote IDs. Live relay uses original request bytes and header values, never the stored identity
Strict replay harness regression tests live in `tests/code_coverage_tests/test_provider_replay_harness.py`. The CircleCI `provider_replay_harness` job runs them alongside the existing legacy harness files with `--noconftest -o pythonpath=tests/e2e`; they need only synthetic HTTP providers and temporary fixture storage
Strict replay harness regression tests live in `tests/code_coverage_tests/test_provider_replay_harness.py`. The CircleCI `provider_replay_harness` job runs them alongside the provider-edge and fixture tests at the root of `tests/e2e_harness/` with `--noconftest -o "pythonpath=tests/e2e tests/e2e_harness"`; they need only synthetic HTTP providers and temporary fixture storage
## MCP OAuth happy path

View file

@ -16,7 +16,7 @@ Requests that differ only by their markers therefore share a canonical identity,
Two different tests never share a recording. A request that reaches the edge without a test segment is forwarded live and never cached, and the edge never names the test from its own process's `PYTEST_CURRENT_TEST`. It used to, and that was wrong whenever the calling test and the serving process differed: the proxy is a separate pod, and under xdist the Claude Code compat matrix registered its shared aliases from every worker, each pointing at that worker's edge, so the router spread one worker's calls across all of them and each call was keyed on whatever test the serving worker was in. Builds 234 and 235 of the e2e pipeline, same commit, credited the same Bedrock request to unrelated tests 92% of the time, which is why that mount never converged
The Claude Code compat cells are not cached. Their aliases are registered once per worker session and shared by every cell, so no call to them belongs to one test, and the matrix exists to prove the real CLI against real providers; `claude_code/conftest.py` registers them with `provider_live=True`. The driver still pins the CLI's config directory, working directory, device id and session id (`_driver_unit_tests/test_request_determinism.py` holds that), so a CLI-driven deployment registered by one test would send stable bytes. Normalizing those values in the key instead would hide a real defect class, since a rule cannot tell a client's own churn from a value a test means to assert on
The Claude Code compat cells are not cached. Their aliases are registered once per worker session and shared by every cell, so no call to them belongs to one test, and the matrix exists to prove the real CLI against real providers; `claude_code/conftest.py` registers them with `provider_live=True`. The driver still pins the CLI's config directory, working directory, device id and session id (`tests/e2e_harness/claude_code/test_request_determinism.py` holds that), so a CLI-driven deployment registered by one test would send stable bytes. Normalizing those values in the key instead would hide a real defect class, since a rule cannot tell a client's own churn from a value a test means to assert on
Provider `Set-Cookie` headers are dropped before validation and never recorded: the edge already withholds them from the proxy, and OpenAI responses always carry Cloudflare bot-management cookies

View file

@ -47,9 +47,8 @@ The per-mode env vars and URL shapes above were captured from a real
docs; if a CLI release changes them, the cells fail with the CLI's own
diagnostic rather than silently testing the wrong wire.
`run_models` and `env` are injection seams for
`_driver_unit_tests/test_passthrough.py`; production callers leave
them unset.
`run_models` and `env` are injection seams for tests; production
callers leave them unset.
"""
from __future__ import annotations

View file

@ -168,10 +168,9 @@ def _manifest_feature_ids() -> FrozenSet[str]:
"""Return the set of feature_ids declared in `manifest.yaml`.
Used as a positive filter so only directories that correspond to a
real matrix row contribute results — utility/support directories
(e.g. `_driver_unit_tests`, `_builder_unit_tests`) are dropped
regardless of naming convention, and the rate-limit summary stays
clean.
real matrix row contribute results — a sibling folder that is not a
matrix row is dropped regardless of naming convention, and the
rate-limit summary stays clean.
Returns an empty set if the manifest is missing or malformed; the
caller treats that as "no path is a feature path", which is the
@ -199,11 +198,11 @@ def _infer_feature_and_provider(node_path: Path) -> Optional[tuple]:
"""Infer (feature_id, provider) from a test file path.
Path shape: tests/e2e/claude_code/<feature_id>/test_<provider>.py
Returns None if the file is not a per-feature test (e.g. unit tests
under `_driver_unit_tests/`), so those don't pollute the matrix
artifact. We positively filter the parent directory against
`manifest.yaml` rather than relying on naming conventions, because
non-feature siblings don't all share an underscore prefix.
Returns None if the file is not a per-feature test, so a sibling that
is not a matrix row never pollutes the matrix artifact. We positively
filter the parent directory against `manifest.yaml` rather than
relying on naming conventions, because non-feature siblings don't all
share an underscore prefix.
"""
name = node_path.name
if not name.startswith("test_") or not name.endswith(".py"):
@ -479,10 +478,9 @@ def pytest_sessionfinish(session, exitstatus):
the rate-limit summary. Single-process runs (no xdist) take the
same code path with a single shard, so behavior is consistent.
Skip when no compat results were collected — this conftest is
loaded for every test under `tests/e2e/claude_code/`, including sibling
unit-test trees (e.g. `_driver_unit_tests/`). Writing an empty
artifact would silently overwrite a real artifact from a prior
Skip when no compat results were collected — a `-k` narrowed run
under `tests/e2e/claude_code/` still reaches this hook. Writing an
empty artifact would silently overwrite a real artifact from a prior
compat-test run on the same checkout.
The xdist controller hits this hook with `_COLLECTOR.items` empty
@ -578,8 +576,8 @@ from claude_code._compat_models import ( # noqa: E402
def _build_control_plane_client(proxy_config: ProxyConfig):
"""Local import of the shared harness so the pure-unit-test tree
under ``_driver_unit_tests/`` etc. never has to pull it in. The
"""Local import of the shared harness so collecting this folder never
pulls it in (nor the env it reads at import) before a cell runs. The
control plane transport is what /model/new lives on; SplitTransport
routes it correctly for both monolithic and split deployments.

View file

@ -378,13 +378,9 @@ curl -fsS "${HEALTH_URL}" >/dev/null \
# ---------------------------------------------------------------------------
RESULTS_JSON="${WORKDIR}/compat-results.json"
# The `_*_unit_tests` ignore is defensive: those harness-only trees are
# markerless (they run without a proxy) and don't feed matrix cells, so
# the cron skips them if/when they land in the suite.
PYTEST_ARGS=(
tests/e2e/claude_code/
--confcutdir=tests/e2e/claude_code
"--ignore-glob=*_unit_tests*"
)
if [[ -n "${PYTEST_K}" ]]; then
log "PYTEST_K set; narrowing to: ${PYTEST_K}"

View file

@ -23,8 +23,8 @@ from coverage_registry.management_cases import case_properties
from e2e_metadata import step_properties, subject_properties
# Hardcoded because the runner image copies tests/e2e/ to /app/e2e, so nothing
# at runtime names this suite's place in the repo. test_junit_properties.py
# fails from a checkout if it moves.
# at runtime names this suite's place in the repo. tests/e2e_harness's
# test_junit_properties.py fails from a checkout if it moves.
SUITE_ROOT = "tests/e2e"

View file

@ -29,7 +29,7 @@ from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta
from lifecycle import ResourceManager
from logging_client import INVALID_UPSTREAM_API_KEY, LoggingClient, first_ok, readiness_details_body
from models import LiteLLMParamsBody
from otel_client import CallTraces, JaegerSpan, JaegerTrace, OtelReader
from otel_client import TTFT_TAG, CallTraces, JaegerSpan, JaegerTrace, OtelReader, one_served_genai_span
from pydantic import BaseModel, ConfigDict, ValidationError
pytestmark = pytest.mark.e2e
@ -138,44 +138,6 @@ def _poll(otel_reader: OtelReader, *, call_id: str, route: str, genai_span: str)
)
def _tag(span: JaegerSpan, key: str) -> str | int | float | bool | None:
for tag in span.tags:
if tag.key == key:
return tag.value
return None
#: The v2 gen-AI span attribute recording time-to-first-token for streamed
#: calls: seconds from the upstream request being issued to the first streamed
#: chunk (stamped only for streaming; added in #32236).
TTFT_TAG = "gen_ai.response.time_to_first_chunk"
#: Jaeger's rendering of a span whose OTEL status is ERROR.
ERROR_STATUS_TAG = "otel.status_code"
def served_genai_spans(trace: JaegerTrace, genai_span: str) -> list[JaegerSpan]:
"""The gen-AI spans for attempts that actually served the request.
The proxy opens one gen-AI span per upstream attempt, so a call the router
retried carries an error span for every failed attempt beside the one that
answered. Only the served attempt streams chunks, so only it records TTFT
or a streaming flag; asserting over the raw span list makes every one of
these tests fail whenever the upstream 429s, 529s, or hands back a stale
credential on the first try."""
return [
span for span in trace.spans if span.operation_name == genai_span and _tag(span, ERROR_STATUS_TAG) != "ERROR"
]
def one_served_genai_span(trace: JaegerTrace, genai_span: str) -> JaegerSpan:
served = served_genai_spans(trace, genai_span)
assert len(served) == 1, (
f"a streamed call must produce exactly ONE served gen-AI span, got {len(served)}; spans: {trace.span_names()}"
)
return served[0]
def _assert_real_ttft(hits: tuple[JaegerTrace, ...], *, genai_span: str) -> None:
"""The enforced behavior: the gen-AI span for the attempt that served the
stream records a TTFT that is a real measurement - present, numeric,
@ -192,7 +154,7 @@ def _assert_real_ttft(hits: tuple[JaegerTrace, ...], *, genai_span: str) -> None
trace = hits[0]
span = one_served_genai_span(trace, genai_span)
value = _tag(span, TTFT_TAG)
value = span.tag(TTFT_TAG)
assert value is not None, (
f"the gen-AI span must record {TTFT_TAG} for a streamed call; "
f"tags present: {sorted(tag.key for tag in span.tags)}"
@ -249,10 +211,10 @@ def _assert_error_span_contract(span: JaegerSpan) -> None:
error.message whose embedded provider error JSON still parses and whose
text also rides the span status description."""
for key, expected in EXPECTED_ERROR_SPAN_ATTRIBUTES.items():
actual = _tag(span, key)
actual = span.tag(key)
assert str(actual) == expected, f"error span attribute {key!r} must be {expected!r}, got {actual!r}"
message = _tag(span, "error.message")
message = span.tag("error.message")
assert isinstance(message, str) and message, "error span must carry a non-empty error.message"
assert "AnthropicException" in message, (
f"error.message must carry the upstream provider exception, got: {message[:200]}"
@ -272,10 +234,10 @@ def _assert_error_span_contract(span: JaegerSpan) -> None:
assert provider_error.error.message.strip(), (
f"the embedded provider error must carry a non-empty message; parsed: {provider_error}"
)
assert _tag(span, "otel.status_description") == message, (
assert span.tag("otel.status_description") == message, (
"the span status description must carry the same untruncated message as error.message"
)
stack = _tag(span, "litellm.provider.error.stack_trace")
stack = span.tag("litellm.provider.error.stack_trace")
assert isinstance(stack, str) and stack, "the error span must carry a non-empty litellm.provider.error.stack_trace"
@ -484,7 +446,7 @@ class TestOtelTraceCompleteness:
_assert_complete_trace(traces, route=route, genai_span=genai_span)
served = one_served_genai_span(traces.hits[0], genai_span)
assert _tag(served, "litellm.request.streaming") is True, (
assert served.tag("litellm.request.streaming") is True, (
"the gen-AI span must record litellm.request.streaming=true; its absence means "
"the stream flag was dropped before the model call"
)
@ -541,7 +503,7 @@ class TestOtelTraceCompleteness:
_assert_complete_trace(traces, route=route, genai_span=genai_span)
served = one_served_genai_span(traces.hits[0], genai_span)
assert _tag(served, "litellm.request.streaming") is True, (
assert served.tag("litellm.request.streaming") is True, (
"the gen-AI span must record litellm.request.streaming=true; its absence means "
"the stream flag was dropped before the model call"
)
@ -804,8 +766,8 @@ class TestOtelTraceCompleteness:
_assert_complete_trace(traces, route=route, genai_span=genai_span)
root = next(span for span in traces.hits[0].spans if not span.references)
assert str(_tag(root, "http.status_code")) == "401", (
f"the SERVER span must record the 401 the client received, got {_tag(root, 'http.status_code')!r}"
assert str(root.tag("http.status_code")) == "401", (
f"the SERVER span must record the 401 the client received, got {root.tag('http.status_code')!r}"
)
genai = next(span for span in traces.hits[0].spans if span.operation_name == genai_span)
_assert_error_span_contract(genai)
@ -866,8 +828,8 @@ class TestOtelTraceCompleteness:
_assert_complete_trace(traces, route=route, genai_span=genai_span)
root = next(span for span in traces.hits[0].spans if not span.references)
assert str(_tag(root, "http.status_code")) == "401", (
f"the SERVER span must record the 401 the client received, got {_tag(root, 'http.status_code')!r}"
assert str(root.tag("http.status_code")) == "401", (
f"the SERVER span must record the 401 the client received, got {root.tag('http.status_code')!r}"
)
genai = next(span for span in traces.hits[0].spans if span.operation_name == genai_span)
_assert_error_span_contract(genai)

View file

@ -37,6 +37,12 @@ from e2e_http import URL, NetworkError, NoBody, Result, Success, get
JAEGER_SERVICE = "litellm"
#: Span tag carrying the request's x-litellm-call-id (stamped on the gen-AI span).
CALL_ID_TAG = "litellm.call_id"
#: The v2 gen-AI span attribute recording time-to-first-token for streamed
#: calls: seconds from the upstream request being issued to the first streamed
#: chunk (stamped only for streaming; added in #32236).
TTFT_TAG = "gen_ai.response.time_to_first_chunk"
#: Jaeger's rendering of a span whose OTEL status is ERROR.
ERROR_STATUS_TAG = "otel.status_code"
class JaegerTag(BaseModel):
@ -72,6 +78,12 @@ class JaegerSpan(BaseModel):
return str(tag.value)
return ""
def tag(self, key: str) -> str | int | float | bool | None:
for entry in self.tags:
if entry.key == key:
return entry.value
return None
class JaegerTrace(BaseModel):
model_config = ConfigDict(extra="ignore", populate_by_name=True)
@ -118,6 +130,26 @@ def root_span(trace: JaegerTrace) -> JaegerSpan | None:
return roots[0] if len(roots) == 1 else None
def served_genai_spans(trace: JaegerTrace, genai_span: str) -> list[JaegerSpan]:
"""The gen-AI spans for attempts that actually served the request.
The proxy opens one gen-AI span per upstream attempt, so a call the router
retried carries an error span for every failed attempt beside the one that
answered. Only the served attempt streams chunks, so only it records TTFT
or a streaming flag; asserting over the raw span list makes every one of
these tests fail whenever the upstream 429s, 529s, or hands back a stale
credential on the first try."""
return [span for span in trace.spans if span.operation_name == genai_span and span.tag(ERROR_STATUS_TAG) != "ERROR"]
def one_served_genai_span(trace: JaegerTrace, genai_span: str) -> JaegerSpan:
served = served_genai_spans(trace, genai_span)
assert len(served) == 1, (
f"a streamed call must produce exactly ONE served gen-AI span, got {len(served)}; spans: {trace.span_names()}"
)
return served[0]
def _follows(trace: JaegerTrace, parent_trace_id: str, parent_span_id: str) -> bool:
root = root_span(trace)
return root is not None and any(

View file

@ -0,0 +1,17 @@
# e2e harness tests
Tests of the harness under `tests/e2e/` (the transport, the clients, the fixture bundle and replay edge, the stack lock, the IdP launcher, the coverage collector, the JUnit properties, the load aggregation helpers and the Claude Code driver), not of the product. They live outside `tests/e2e/` because the Buildkite e2e run copies that folder into the runner image and runs every test in it, so a harness test in there counts as a product test in the nightly numbers. Nothing here needs a proxy, provider keys or the network
The layout mirrors `tests/e2e/`: `test_e2e_http.py` covers `tests/e2e/e2e_http.py`, `logging/test_datadog_reader.py` covers `tests/e2e/logging/datadog_reader.py`, and `claude_code/` covers the driver, builder, probe and version resolver. Put a new harness test under the folder that mirrors the suite folder whose module it covers
Run them from the repo root. `e2e_config` reads `LITELLM_MASTER_KEY` at import and any value will do, the CI lane sets a dummy:
```bash
LITELLM_MASTER_KEY=sk-harness uv run pytest tests/e2e_harness
```
`pytest.ini` here puts `tests/e2e` and the suite folders whose modules are under test on the path, so imports look exactly as they do inside the suite (`from e2e_http import ...`, `from batch_cleanup import ...`). `claude_code/test_request_determinism.py` drives the real `claude` CLI; deselect it with `-m "not cli_determinism"` when the CLI is not installed
Rules: no `e2e` marker and no `@meta`, since nothing here drives the proxy; `@pytest.mark.covers` only where the test proves the collector or the JUnit properties read it; inputs via arguments or env vars (setting an env var through pytest's `monkeypatch` fixture is fine, patching a function, class or module is not); and the same typing bar as the suite, `make lint-e2e-basedpyright` covers this folder and allows zero errors. The raw HTTP client ban (`tests/code_coverage_tests/check_e2e_no_raw_requests.py`) applies here too
CI: the `lint` job in `.github/workflows/test-linting.yml` runs this folder whenever anything under `tests/e2e/` (except `ui/`) or `tests/e2e_harness/` changes, with the `claude` CLI installed. The CircleCI `provider_replay_harness` job also runs the provider-edge and fixture tests at the root of this folder next to `tests/code_coverage_tests/test_provider_replay_harness.py`, which imports helpers from `test_provider_edge.py`

View file

@ -1,17 +1,15 @@
"""Harness coverage for the gen-AI span selection in `test_otel_trace_e2e`.
"""Harness coverage for the gen-AI span selection `logging/test_otel_trace_e2e` relies on.
Carries no `e2e` marker: this exercises the selection helper itself against
Jaeger-shaped payloads, so it runs whether or not a proxy is up. The live
assertions it protects are expensive to reproduce (they need an upstream that
fails the first attempt), which is exactly why the helper is worth pinning
here.
This exercises the selection helper itself against Jaeger-shaped payloads, so it
runs without a proxy. The live assertions it protects are expensive to reproduce
(they need an upstream that fails the first attempt), which is exactly why the
helper is worth pinning here.
"""
from __future__ import annotations
import pytest
from otel_client import JaegerTrace
from test_otel_trace_e2e import TTFT_TAG, one_served_genai_span, served_genai_spans
from otel_client import TTFT_TAG, JaegerTrace, one_served_genai_span, served_genai_spans
GENAI_SPAN = "chat claude-haiku-4-5"

View file

@ -0,0 +1,9 @@
[pytest]
# Tests of the tests/e2e harness itself. They import harness modules by bare name
# (`from e2e_http import ...`, `from batch_cleanup import ...`) exactly as the suites
# do, so tests/e2e and each suite folder that owns a module under test go on the path.
addopts = --strict-markers --strict-config -p no:cacheprovider
pythonpath = ../e2e ../e2e/batches ../e2e/guardrails ../e2e/load ../e2e/logging
markers =
covers(cell_id, *, exercised_on=()): coverage-registry cell(s) a test covers; exercised here only to prove the collector and the JUnit properties read it
cli_determinism: drives the real claude CLI for several seconds

View file

@ -4,6 +4,7 @@ these carry no `e2e` marker and run everywhere."""
from __future__ import annotations
import inspect
import os
import signal
import socket
@ -19,6 +20,7 @@ from queue import SimpleQueue
from threading import Thread
from typing import Final, Literal
import idp
import pytest
from e2e_http import ExternalWrite
from idp import (
@ -35,6 +37,7 @@ from idp import (
keycloak_from_env,
)
IDP_SCRIPT: Final = inspect.getfile(idp)
_REALM: Final = Keycloak(
base_url="http://keycloak:8080", realm="litellm-e2e", admin_username="admin", admin_password="pw"
)
@ -175,7 +178,7 @@ def test_oidc_launcher_removes_client_on_exit_and_termination(
with subprocess.Popen(
[
sys.executable,
str(Path(__file__).with_name("idp.py")),
IDP_SCRIPT,
"http://127.0.0.1:9999",
sys.executable,
"-c",

View file

@ -10,8 +10,10 @@ rollups and, for ``source``, the status page's per-test links to GitHub.
from __future__ import annotations
import inspect
from pathlib import Path
import junit_properties
import pytest
from junit_properties import (
SUITE_ROOT,
@ -97,13 +99,16 @@ class TestResultProperties:
def test_every_test_carries_package_covers_and_source(self, request: pytest.FixtureRequest) -> None:
"""Read off this test's own collected Item, so the nodeid and location are
whatever pytest reports for the launch shape in use, and the marker is added
at run time so the coverage registry's collect-only pass never sees it."""
at run time so the coverage registry's collect-only pass never sees it. The
source re-roots the location under the suite root as it would for a suite
file: the constant is hardcoded, not looked up, so a file outside the suite
gets the same treatment."""
test = type(self).test_every_test_carries_package_covers_and_source
request.applymarker(pytest.mark.covers("LOG-1", "LOG-2"))
assert result_properties(collected_item(request, test.__name__)) == (
("package", "root"),
("covers", "LOG-1,LOG-2"),
("source", f"tests/e2e/test_junit_properties.py:{test.__code__.co_firstlineno}"),
("source", f"{SUITE_ROOT}/{Path(__file__).name}:{test.__code__.co_firstlineno}"),
)
def test_attach_is_idempotent(self, request: pytest.FixtureRequest) -> None:
@ -116,14 +121,15 @@ class TestResultProperties:
class TestSuiteRoot:
def test_suite_root_names_this_file_s_real_home(self) -> None:
def test_suite_root_names_the_harness_s_real_home(self) -> None:
"""SUITE_ROOT is hardcoded because the runner image has no repo to read it
from. Where there IS a checkout, prove the constant still points at us --
otherwise a moved tests/e2e/ ships links that 404."""
from. Where there IS a checkout, prove the constant still points at the
harness -- otherwise a moved tests/e2e/ ships links that 404."""
root = repo_root()
if root is None:
pytest.skip("no checkout above this file (the runner image copies tests/e2e/ to /app/e2e)")
assert (root / SUITE_ROOT / Path(__file__).name).resolve() == Path(__file__).resolve()
pytest.skip("no checkout above this file")
harness_home = Path(inspect.getfile(junit_properties)).resolve()
assert (root / SUITE_ROOT / "junit_properties.py").resolve() == harness_home
class TestDedupeCovers:

View file

@ -5,6 +5,7 @@ queues behind it instead of starving it."""
from __future__ import annotations
import fcntl
import inspect
import os
import subprocess
import sys
@ -14,10 +15,9 @@ from pathlib import Path
from typing import Final
import pytest
import stack_lock
from stack_lock import STACK_DIGEST
HARNESS_DIR: Final = Path(__file__).resolve().parent
HARNESS_DIR: Final = Path(inspect.getfile(stack_lock)).resolve().parent
DEADLINE_SECONDS: Final = 30.0
SETTLE_SECONDS: Final = 0.5
HOLDER_SCRIPT: Final = """
@ -89,7 +89,7 @@ def _start_holder(held: ExitStack, tmp_path: Path, name: str, mode: str) -> subp
def test_readers_share_exclusive_waits_and_a_waiting_exclusive_beats_later_readers(tmp_path: Path) -> None:
lock_dir: Final = tmp_path / f"litellm-e2e-stack-{STACK_DIGEST}"
lock_dir: Final = tmp_path / f"litellm-e2e-stack-{stack_lock.STACK_DIGEST}"
lock_dir.mkdir()
log_path: Final = tmp_path / "events"
with ExitStack() as held:

View file

@ -62,6 +62,8 @@ CI = [".github/workflows/test-litellm-ui-unit.yml"]
("provider-harness", ["tests/e2e/provider_cache.py"], "run"),
("provider-harness", ["tests/e2e/conftest.py"], "run"),
("provider-harness", ["tests/e2e/e2e_http.py"], "run"),
("provider-harness", ["tests/e2e_harness/test_provider_edge.py"], "run"),
("provider-harness", ["tests/e2e_harness/logging/test_datadog_reader.py"], "skip"),
("provider-harness", ["tests/code_coverage_tests/test_provider_cache.py"], "run"),
("provider-harness", ["tests/code_coverage_tests/test_provider_replay_harness.py"], "run"),
("provider-harness", [".circleci/config.yml"], "run"),