test(e2e): move matrix data freshness checks to collection time

Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
kerry 2026-09-17 01:22:01 +00:00
parent 072b32baf2
commit 1de633ac36
4 changed files with 53 additions and 62 deletions

View file

@ -21,7 +21,7 @@ Each subdirectory under `tests/e2e/` is one suite, scoped to an endpoint family
- `load/` - performance-category tests, kept OUT of the main suite: throughput/load SLO tests are a different testing category from functional e2e (variance-driven, historically flaky) and live outside this suite until re-implemented as their own pipeline (LIT-5163); do not add a live load test that runs in the default collection. What lives here: the weekly session-anomaly test (`test_weekly_session_anomaly_e2e.py`, Claude Code-shaped multi-turn sessions against real providers with ceilings on error rate, cache read/write, turn time, and spend; marked `weekly` and deselected unless `E2E_WEEKLY_ANOMALY` is set, driven by `.github/workflows/weekly_load_anomaly.yml`), the Redis chaos test (`test_redis_chaos_e2e.py`, locust load against mock deployments split round robin over `/chat/completions` and `/v1/messages`, one endpoint per simulated user, with `CLIENT PAUSE ALL` on the proxy's Redis mid-run to simulate it being down outright, asserting zero failed requests on every endpoint, budgeting RSS and CPU-per-request as ratios against the same run's healthy phase, and holding p50/p90/p99 latency and log-bytes-per-request to flat ceilings (a ratio cannot bound those two: an open breaker skips Redis instead of waiting on it, so the chaos phase can measure cheaper than baseline while still being far slower than a user should see); needs a proxy booted from `gateway/redis_chaos_ci_config.yml` on the same host with `E2E_PROXY_PID` and `E2E_PROXY_LOG` set, marked `redis_chaos`, deselected unless `E2E_REDIS_CHAOS` is set and excluded from the per-PR selector like the rest of `load/`, driven by `.github/workflows/test-e2e-redis-chaos.yml` and by the Buildkite `e2e-redis-chaos` step in project-releaser, which runs the proxy, Postgres and Valkey co-located with pytest in one pod and sets the opt-in), and markerless harness unit tests for the locust, process-usage, and session-anomaly aggregation logic
- `other/` - the holding-pen suite for the `other.*` registry cluster with no home of its own yet: the master-key auth gate, JWT auth (access tokens issued by a real Keycloak realm, `idp.py` plus `idp_realm.json`, whose JWKS the proxy's `JWT_PUBLIC_KEY_URL` points at; see CONTRIBUTING.md for the start command and config block), and the process-lifecycle health probes (liveness, public readiness, authenticated readiness diagnostics). Promote a cluster out once it is large/stable enough for its own suite
- `gateway/` - proxy configuration only (`litellm-config.yml`); no tests
- `cost_calculation/` - cost accounting against a dedicated proxy whose whole model cost map is the test-owned `tests/e2e/cost_map.json` (loaded via `LITELLM_MODEL_COST_MAP_URL`), with provider calls answered by the scripted-provider sidecar in `scripted_provider.py`; every cost-map entry is a deployment and the cases plus asserted goldens are data in `cases.json` and `expected.json` (regenerate with `generate_expected.py`), deselected unless `E2E_COST_MAP_STACK` is set, driven by the Buildkite `e2e-cost-calculation` step in project-releaser, which runs a proxy booted from `gateway/cost_calculation_ci_config.yml`, Postgres and the scripted provider co-located with pytest in one pod and sets the opt-in
- `cost_calculation/` - cost accounting against a dedicated proxy whose whole model cost map is the test-owned `tests/e2e/cost_map.json` (loaded via `LITELLM_MODEL_COST_MAP_URL`), with provider calls answered by the scripted-provider sidecar in `scripted_provider.py`; every cost-map entry is a deployment and the cases plus asserted goldens are data in `cases.json` and `expected.json` (regenerate with `generate_expected.py`; `cost_matrix.matrix_data_errors()` runs at collection time so a stale key set fails the suite's collection loudly), deselected unless `E2E_COST_MAP_STACK` is set, driven by the Buildkite `e2e-cost-calculation` step in project-releaser, which runs a proxy booted from `gateway/cost_calculation_ci_config.yml`, Postgres and the scripted provider co-located with pytest in one pod and sets the opt-in
- `claude_code/` - the Claude Code compatibility matrix: drives the real `claude` CLI (and HTTP probes) against a proxy for each feature x provider cell, reporting tagged-union outcomes via the `compat_result` fixture; ships its own driver/builder/publisher plus `_*_unit_tests/` trees. The HTTP probes ride the shared transport (`ProxyClient.count_tokens` / `ProxyClient.messages`); the CLI-driving path stays bespoke
- `ui/` - the Admin UI browser suite: Playwright in TypeScript, driving the dashboard served by a live proxy on port 4000 (seeded postgres + mock LLM upstream; see its `run_e2e.sh`). It is a self-contained npm package with its own lockfile and does not use the Python harness, pytest markers, or the shared transport; the Python rules in this file (typed models, `Result` unions, basedpyright zero-error gate) do not apply inside it. Its only Python file, `fixtures/mock_llm_server/server.py`, is excluded from the e2e basedpyright gate via the root `pyrightconfig.json`

View file

@ -435,3 +435,51 @@ EXPECTED: Final[Mapping[str, ExpectedCell]] = MappingProxyType(
def expected_key(model: FrontierModel, case: Case) -> str:
return f"{model.map_key}|{case.name}"
def matrix_data_errors() -> tuple[str, ...]:
"""Freshness findings for the data files, as human-readable strings.
Called at collection time by the e2e suite; also usable from
generate_expected.py's context without importing pytest.
"""
derived: Final = {
expected_key(model, case)
for model in FRONTIER_MODELS
for case in cases_for(model)
if case.exact_spend
}
golden: Final = set(EXPECTED)
unknown_deployments: Final = sorted(
spec.map_key for spec in CASES_FILE.deployments if spec.map_key not in COST_MAP
)
unknown_rates: Final = sorted(
{field for case in CASES for field in case.requires_rates} - set(CostMapEntry.model_fields)
)
input_rates: Final = tuple(entry.input_cost_per_token for entry in COST_MAP.values())
findings: Final = (
(
"expected.json is out of sync with the derived matrix; run "
"uv run python tests/e2e/cost_calculation/generate_expected.py "
f"(missing: {sorted(derived - golden)}; stale: {sorted(golden - derived)})"
)
if derived != golden
else None,
(
f"deployments entries name map keys absent from cost_map.json: {unknown_deployments}"
if unknown_deployments
else None
),
(
f"requires_rates names that are not CostMapEntry fields: {unknown_rates}"
if unknown_rates
else None
),
(
"two cost_map entries share input_cost_per_token; the suite relies on "
"distinct rates so a wrong-model bill can never coincidentally match"
if len(input_rates) != len(set(input_rates))
else None
),
)
return tuple(finding for finding in findings if finding is not None)

View file

@ -1,61 +0,0 @@
"""Freshness checks for the cost suite's data files; markerless, so it runs on
any pytest invocation of the folder without the stack. expected.json is the
oracle: these tests check its key set against the derived matrix, never its
values (the generator proposes, the file decides)."""
from __future__ import annotations
from typing import Final
import pytest
from cost_matrix import (
CASES,
CASES_FILE,
COST_MAP,
EXPECTED,
FRONTIER_MODELS,
CostMapEntry,
cases_for,
expected_key,
)
def test_expected_keys_match_derived_exact_cells() -> None:
derived: Final = {
expected_key(model, case)
for model in FRONTIER_MODELS
for case in cases_for(model)
if case.exact_spend
}
golden: Final = set(EXPECTED)
if derived != golden:
missing: Final = sorted(derived - golden)
stale: Final = sorted(golden - derived)
pytest.fail(
"expected.json is out of sync with the derived matrix; run "
"uv run python tests/e2e/cost_calculation/generate_expected.py "
f"(missing: {missing}; stale: {stale})"
)
def test_deployments_reference_existing_map_keys() -> None:
unknown: Final = sorted(
spec.map_key for spec in CASES_FILE.deployments if spec.map_key not in COST_MAP
)
assert not unknown, f"deployments entries name map keys absent from cost_map.json: {unknown}"
def test_requires_rates_are_cost_map_fields() -> None:
fields: Final = set(CostMapEntry.model_fields)
unknown: Final = sorted(
{field for case in CASES for field in case.requires_rates} - fields
)
assert not unknown, f"requires_rates names that are not CostMapEntry fields: {unknown}"
def test_no_two_entries_share_input_rate() -> None:
rates: Final = tuple(entry.input_cost_per_token for entry in COST_MAP.values())
assert len(rates) == len(set(rates)), (
"two cost_map entries share input_cost_per_token; the suite relies on "
"distinct rates so a wrong-model bill can never coincidentally match"
)

View file

@ -22,6 +22,7 @@ from cost_matrix import (
FrontierModel,
cases_for,
expected_key,
matrix_data_errors,
recount_cost,
)
from e2e_config import unique_marker
@ -39,6 +40,9 @@ from models import (
pytestmark: Final = [pytest.mark.e2e, pytest.mark.cost_map_stack] # mutable-ok: pytest only accepts a list for pytestmark
if _data_errors := matrix_data_errors():
raise ValueError("\n".join(_data_errors))
_MATRIX: Final[tuple[tuple[FrontierModel, Case], ...]] = tuple(
(model, case) for model in FRONTIER_MODELS for case in cases_for(model)
)