From 4d042a9ae4461554a277afb4588c68fbcf8e5934 Mon Sep 17 00:00:00 2001 From: Ishaan Jaff Date: Wed, 8 Jul 2026 12:16:43 -0700 Subject: [PATCH] test: reorganize e2e coverage registry --- .github/workflows/test-code-quality.yml | 5 + tests/e2e/CLAUDE.md | 124 ++---- .../access_control/test_access_control_e2e.py | 26 +- tests/e2e/budgets/test_budget_crud_e2e.py | 36 +- .../budgets/test_budget_enforcement_e2e.py | 15 +- tests/e2e/budgets/test_budget_fallback_e2e.py | 22 +- tests/e2e/budgets/test_budget_reset_e2e.py | 10 +- .../e2e/budgets/test_model_max_budget_e2e.py | 16 +- .../budgets/test_multi_window_budget_e2e.py | 20 +- tests/e2e/budgets/test_soft_budget_e2e.py | 14 +- .../budgets/test_spend_counter_reseed_e2e.py | 18 +- tests/e2e/budgets/test_tag_budget_e2e.py | 16 +- .../budgets/test_team_member_budget_e2e.py | 42 +- .../test_team_member_budget_reset_e2e.py | 37 +- .../test_team_multi_window_budget_e2e.py | 26 +- tests/e2e/conftest.py | 5 +- tests/e2e/coverage_registry/README.md | 131 +++--- tests/e2e/coverage_registry/__init__.py | 9 +- .../coverage_registry/check_coverage_sync.py | 80 ++++ tests/e2e/coverage_registry/collector.py | 384 +++++++++++++----- tests/e2e/coverage_registry/guardrail.yaml | 29 -- .../coverage_registry/llm_conversational.yaml | 54 --- .../llm_nonconversational.yaml | 45 -- tests/e2e/coverage_registry/logging.yaml | 25 -- tests/e2e/coverage_registry/mcp.yaml | 113 ------ tests/e2e/coverage_registry/mgmt.yaml | 68 ---- tests/e2e/coverage_registry/other.yaml | 28 -- tests/e2e/coverage_registry/registry.py | 26 -- tests/e2e/coverage_registry/reliability.yaml | 30 -- tests/e2e/coverage_registry/schema.py | 322 ++++++++------- tests/e2e/coverage_registry/test_collector.py | 244 ++++------- .../{ => llm_translation}/batches/COVERAGE.md | 0 .../batches/batch_client.py | 0 .../batches/capabilities.py | 4 +- .../{ => llm_translation}/batches/conftest.py | 4 +- .../batches/test_batches_e2e.py | 36 +- .../realtime/REALTIME_COVERAGE_MATRIX.md | 2 +- .../realtime/conftest.py | 2 +- .../fixtures/weather_question_24k.wav | Bin .../realtime/pipecat_service.py | 0 .../realtime/realtime_client.py | 0 .../realtime/test_realtime_e2e.py | 12 +- .../test_realtime_pipecat_audio_e2e.py | 16 +- .../realtime/test_realtime_pipecat_e2e.py | 16 +- .../llm_translation/test_audio_speech_e2e.py | 16 +- .../test_chat_completions_regression_e2e.py | 16 +- .../test_custom_pricing_e2e.py | 31 +- .../test_deepseek_reasoning_e2e.py | 10 +- .../test_embeddings_endpoint_e2e.py | 23 +- .../test_image_generation_e2e.py | 16 +- .../e2e/llm_translation/test_messages_e2e.py | 17 +- .../e2e/llm_translation/test_ocr_rust_e2e.py | 21 +- .../llm_translation/test_passthrough_e2e.py | 14 +- .../test_provider_features_e2e.py | 17 +- tests/e2e/llm_translation/test_rerank_e2e.py | 20 +- .../e2e/llm_translation/test_responses_e2e.py | 18 +- .../test_vertex_passthrough_e2e.py | 22 +- .../test_prometheus_cardinality_e2e.py | 23 +- .../test_key_models_dropdown_e2e.py | 111 +++-- tests/e2e/management/test_management_e2e.py | 18 +- tests/e2e/pytest.ini | 1 + tests/e2e/rate_limits/README.md | 13 + tests/e2e/spend_tracking/test_spend_routes.py | 10 +- .../spend_tracking/test_spend_tracking_e2e.py | 55 ++- tests/e2e/test_e2e_gateway.py | 12 +- tests/e2e/test_lifecycle.py | 7 + tests/e2e/test_transport.py | 21 +- 67 files changed, 1401 insertions(+), 1223 deletions(-) create mode 100644 tests/e2e/coverage_registry/check_coverage_sync.py delete mode 100644 tests/e2e/coverage_registry/guardrail.yaml delete mode 100644 tests/e2e/coverage_registry/llm_conversational.yaml delete mode 100644 tests/e2e/coverage_registry/llm_nonconversational.yaml delete mode 100644 tests/e2e/coverage_registry/logging.yaml delete mode 100644 tests/e2e/coverage_registry/mcp.yaml delete mode 100644 tests/e2e/coverage_registry/mgmt.yaml delete mode 100644 tests/e2e/coverage_registry/other.yaml delete mode 100644 tests/e2e/coverage_registry/registry.py delete mode 100644 tests/e2e/coverage_registry/reliability.yaml rename tests/e2e/{ => llm_translation}/batches/COVERAGE.md (100%) rename tests/e2e/{ => llm_translation}/batches/batch_client.py (100%) rename tests/e2e/{ => llm_translation}/batches/capabilities.py (98%) rename tests/e2e/{ => llm_translation}/batches/conftest.py (91%) rename tests/e2e/{ => llm_translation}/batches/test_batches_e2e.py (94%) rename tests/e2e/{ => llm_translation}/realtime/REALTIME_COVERAGE_MATRIX.md (97%) rename tests/e2e/{ => llm_translation}/realtime/conftest.py (86%) rename tests/e2e/{ => llm_translation}/realtime/fixtures/weather_question_24k.wav (100%) rename tests/e2e/{ => llm_translation}/realtime/pipecat_service.py (100%) rename tests/e2e/{ => llm_translation}/realtime/realtime_client.py (100%) rename tests/e2e/{ => llm_translation}/realtime/test_realtime_e2e.py (94%) rename tests/e2e/{ => llm_translation}/realtime/test_realtime_pipecat_audio_e2e.py (97%) rename tests/e2e/{ => llm_translation}/realtime/test_realtime_pipecat_e2e.py (92%) create mode 100644 tests/e2e/rate_limits/README.md diff --git a/.github/workflows/test-code-quality.yml b/.github/workflows/test-code-quality.yml index 872a1799d98..28c91f08845 100644 --- a/.github/workflows/test-code-quality.yml +++ b/.github/workflows/test-code-quality.yml @@ -79,6 +79,11 @@ jobs: - name: code_qa_check_tests run: uv run --no-sync python ./tests/code_coverage_tests/code_qa_check_tests.py + - name: e2e_coverage_registry_sync + run: | + cd tests/e2e + PYTHONPATH=. uv run --no-sync python -m coverage_registry.check_coverage_sync + - name: check_get_model_cost_key_performance run: uv run --no-sync python ./tests/code_coverage_tests/check_get_model_cost_key_performance.py diff --git a/tests/e2e/CLAUDE.md b/tests/e2e/CLAUDE.md index 502c881c764..6ab501e5199 100644 --- a/tests/e2e/CLAUDE.md +++ b/tests/e2e/CLAUDE.md @@ -6,19 +6,23 @@ Code-style rules for writing tests under `tests/e2e/`. The harness already encod Each subdirectory under `tests/e2e/` is one suite, scoped to an endpoint family or behavior area. If you add a new folder, you must add a line here describing what kind of tests belong in it, so the layout stays self-describing. `gateway/` is the exception: it holds proxy configuration only and never tests -- `llm_translation/` - LLM endpoint and provider-translation behavior: passthrough, custom pricing, OCR, and the non-chat inference endpoints (`/v1/responses`, `/v1/messages`, `/embeddings`, `/v1/rerank`, `/v1/audio/speech`, `/v1/images/generations`), each against a deployment the test creates via `/model/new` and deletes on teardown +- `llm_translation/` - LLM endpoint and provider-translation behavior: passthrough, custom pricing, OCR, realtime, batches, and every LLM data-plane endpoint (`/chat/completions`, `/v1/responses`, `/v1/messages`, `/embeddings`, `/v1/rerank`, `/v1/audio/speech`, `/v1/images/generations`, `/v1/batches`, `/v1/realtime`, etc.), each against a deployment the test creates via `/model/new` and deletes on teardown where applicable. Any test whose primary subject is an LLM endpoint belongs here, even if the endpoint uses websockets, files, provider passthrough, or a non-chat modality. - `access_control/` - the gateway's authorization and error-shape contract: per-key model allow-lists, route-group permissions (`allowed_routes`), and unknown-model validation -- `embeddings/` - the `/embeddings` endpoint across providers -- `batches/` - the `/batches` endpoint (placeholder until the first test lands) -- `realtime/` - realtime websocket sessions, including the pipecat audio path - `budgets/` - budget definition, enforcement, and reset windows (key, team, tag, soft, multi-window) +- `rate_limits/` - rate-limit enforcement behavior across keys, teams, models, tags, and endpoint families; endpoint-specific LLM translation assertions still live under `llm_translation/` - `spend_tracking/` - spend logging and cost attribution on `/spend/*` - `management/` - key/team/user/organization management routes: create/update/delete persistence via the info routes, team membership, and llm-only-key route denials; also the dashboard UI behavior on top of them, driven through the proxy-served UI at /ui with playwright (optional dep behind importorskip) +- `mcp/` - MCP protocol behavior and coverage rows: tool/resource/prompt operations, auth-family handling, and MCP-specific error shapes - `logging/` - logging-integration delivery (datadog and friends) +- `reliability/` - routing and reliability behavior (fallbacks, retries, cooldowns, routing strategies) +- `other/` - temporary holding area for coverage rows that do not yet have a stable suite owner; promote rows out of here when a module boundary becomes clear - `security/` - secret handling and log-leak protection -- `router/` - routing and reliability behavior (rate limits, fallbacks, cooldowns) - `gateway/` - proxy configuration only (`litellm-config.yml`); no tests +Do not add new top-level folders for LLM data-plane endpoints. Put endpoint-specific +LLM suites under `llm_translation/` and create a subfolder there only when the +endpoint needs its own client, fixtures, or coverage matrix. + ## Lay the pattern down in a class Keep the cases for one feature inside a class so the file reads as a spec for how that feature behaves in production. The class name says what is under test; each method is one behavior. Think of it as documenting the contract, with the rough intent being @@ -59,94 +63,44 @@ Mark live tests with `@pytest.mark.e2e` (on the class or the module). Pure cover The harness is fully typed and new code must not add `Any` or widen the basedpyright budgets. When a response field is untyped, model it in `models.py` (just the fields you read) and let pydantic validate it, rather than threading a `dict` or `Any` through the test -## Coverage registry +## Coverage metadata -The set of tests we want is a registry checked into this repo, one row per behavior; that file is the definition of done and the denominator. Each e2e test declares what it covers with `@pytest.mark.covers("...")`, and a small collector diffs the registry against the tests and ships coverage to the existing Grafana. No Allure, no new dependencies +Every collected pytest under `tests/e2e` must declare the e2e surface it covers with `@pytest.mark.e2e_coverage(...)`. The marker is the source of truth; there are no coverage YAML files. Use a module-level `pytestmark` when all tests in a file cover the same surface, or add per-test markers when a file mixes endpoints/providers. -Coverage is organized as module > feature > test. Dashboard modules are `Core LLMs`, `Non-Core LLMs`, `MCPs`, `Management/UI`, `Reliability & Performance`, `Logging & Guardrails`, and `Other`. The Loki stdout formatter maps those display modules to log-safe labels (`core_llms`, `non_core_llms`, `mcp`, `management_ui`, `reliability_performance`, `logging_guardrails`, and `other`) without changing JSON or Prometheus labels. A feature is either an endpoint (`/chat/completions`) or a behavior (fallbacks, rate limits; config-driven, with no route of its own). A cell reads like `llm.chat_completions.bedrock_converse.tool_use.stream.works` - -The metric is coverage: the share of registry rows that have a passing covering test, reported to Grafana per module so a gap surfaces as an uncovered row rather than a silent absence - -Tests do not declare a dashboard module directly. They only declare the registry cell id with `@pytest.mark.covers("...")`; the registry row decides the module, tier, endpoint, and dashboard rollup. Run `python -m coverage_registry.collector --strict` when you want CI to reject unknown marker ids. Add `--fail-on-collection-errors` when the job should also fail on pytest collection errors. - -### Naming grammar per module - -LLMs - endpoint features (subject = the route), seeded from the Claude Code compat matrix. `chat_completions`, `messages`, and `responses` roll up to `Core LLMs`. Other LLM endpoints, including `batches` and `realtime`, roll up to `Non-Core LLMs`. - -``` -llm..... - endpoint : chat_completions | messages | responses | embeddings | batches | files - | rerank | images_generations | audio_speech | audio_transcriptions | moderations - | realtime - route : openai | azure_openai | anthropic | bedrock_converse | vertex | azure_foundry - | cohere | together_ai - (vocab varies per endpoint; messages is anthropic-format only) - capability : basic | tool_use | prompt_cache_5m | vision | thinking | structured_output - | service_tier - streaming : stream | nonstream (omit where n/a) - assertion : works | cost_logged - label (not in id): model = haiku-4.5 | sonnet-4.6 | opus-4.7 | gpt-* - e.g. llm.chat_completions.bedrock_converse.tool_use.stream.works - llm.messages.anthropic.prompt_cache_1h.nonstream.cache_hit +```python +pytestmark = [ + pytest.mark.e2e, + pytest.mark.e2e_coverage( + module="core_llms", + endpoint="/chat/completions", + provider="openai", + params=["tools", "streaming"], + ), +] ``` -Management / UI - endpoint features (surface tag: api | ui) +Required fields: -``` -mgmt.. - endpoint : key.generate | key.update | key.delete | team.new | user.new - | budget.new | model.add | ... (one per management route) - assertion : persists | member_forbidden | admin_only | happy_path - e.g. mgmt.key.generate.persists (surface=api) - mgmt.key.generate.happy_path (surface=ui) +- `module`: one of `core_llms`, `non_core_llms`, `access_control`, `budgets`, `spend_tracking`, `management`, `mcp`, `rate_limits`, `reliability`, `logging`, `guardrails`, or `other`. +- `endpoint`: a known endpoint/surface from `coverage_registry/schema.py`, such as `/chat/completions`, `/v1/messages`, `/v1/batches`, `/budget/*`, or `/spend/*`. +- `provider`: a known provider/integration from `coverage_registry/schema.py`, such as `proxy`, `openai`, `anthropic`, `vertex_ai`, or `multiple`. +- `params`: one or more explicit lowercase parameter/behavior names, such as `tools`, `streaming`, `key_rpm_limit`, `budget_enforcement`, or `spend_routes`. + +The collector reads these markers with pytest collect-only and reports unique endpoint x provider x parameter units plus test counts by module. Run this before opening or updating a PR: + +```bash +cd tests/e2e && PYTHONPATH=. python -m coverage_registry.check_coverage_sync ``` -MCPs - endpoint features with the protocol op as the variant +The sync check fails if any collected test is missing `e2e_coverage`, uses an unknown module/endpoint/provider, has empty params, or has pytest collection errors. -``` -mcp... - operation : list_tools | call_tool | list_resources | read_resource | list_prompts | get_prompt - auth_family : none | api_key | bearer | oauth - assertion : succeeds | denied_without_permission - e.g. mcp.call_tool.oauth.succeeds -``` +### Marker value grammar -Reliability & Performance - behavior features (no route; endpoint is exercised_on) +Use stable, dashboard-safe values. Do not include spaces or prose in marker fields. If a new endpoint or provider is legitimate, add it to `coverage_registry/schema.py` in the same PR so the CI check teaches the dashboard about it. -``` -reliability... - behavior : fallback | retry | cooldown | timeout | ratelimit | routing | cache | circuit_breaker | perf - variant : 5xx | context_window | content_policy | 429 | timeout - simple_shuffle | usage_based | latency_based | cost_based | least_busy - latency | throughput (perf only; SLO/threshold assertion, not binary) - assertion : routes_to_fallback | succeeds_within_retries | picks_under_tpm | returns_cached - | trips_then_recovers | under_slo - e.g. reliability.fallback.context_window.routes_to_fallback exercised_on=[chat_completions] - reliability.ratelimit.rpm.blocks_over_limit exercised_on=[chat_completions, messages] -``` +- Core LLM endpoint tests: `module="core_llms"`, endpoints `/chat/completions`, `/v1/messages`, or `/v1/responses`. +- Other LLM endpoint tests: `module="non_core_llms"`, endpoints such as `/v1/batches`, `/v1/realtime`, `/v1/embeddings`, `/v1/audio/speech`, `/v1/images/generations`, or `/rerank`. +- Proxy behavior tests: use the owning module (`budgets`, `rate_limits`, `access_control`, `spend_tracking`, `management`) and the route family being exercised. +- Harness/tooling tests: use `module="other"` or `module="reliability"` with endpoint `e2e_harness` or `coverage_registry`. -Logging & Guardrails - behavior features (config-driven; endpoint is exercised_on) - -``` -logging... - integration : langfuse | s3 | otel | prometheus | datadog | ... - event : success | failure | stream - assertion : logs_spend | writes_object | exports_metric - e.g. logging.langfuse.success.logs_spend exercised_on=[chat_completions] - -guardrail... - provider : presidio | lakera | bedrock | aporia | ... - hook_point : pre_call | post_call | during | logging_only - assertion : blocks | masks | allows - e.g. guardrail.presidio.pre_call.masks exercised_on=[chat_completions] -``` - -Other - holding pen (endpoint or behavior) - -``` -other... - area : auth | lifecycle | config | ... - rule : audited periodically; a cluster here promotes to a new component - e.g. other.auth.jwt.valid_token_allows - other.lifecycle.readiness.reports_db -``` +Prefer precise params like `key_rpm_limit`, `budget_enforcement`, `prompt_cache`, `tool_calling`, or `spend_routes` over generic params like `works`. diff --git a/tests/e2e/access_control/test_access_control_e2e.py b/tests/e2e/access_control/test_access_control_e2e.py index ce649fa2400..1e17ea94dae 100644 --- a/tests/e2e/access_control/test_access_control_e2e.py +++ b/tests/e2e/access_control/test_access_control_e2e.py @@ -25,7 +25,15 @@ from access_control_client import ( from e2e_config import unique_marker from lifecycle import ResourceManager -pytestmark = pytest.mark.e2e +pytestmark = [ + pytest.mark.e2e, + pytest.mark.e2e_coverage( + module="access_control", + endpoint="/chat/completions", + provider="proxy", + params=["model_access", "route_permission", "unknown_model"], + ), +] ALLOWED_MODEL = "gemini-2.5-flash" DISALLOWED_MODEL = "gpt-5.5" @@ -51,9 +59,9 @@ class TestAccessControl: f"key limited to {ALLOWED_MODEL!r} calling {DISALLOWED_MODEL!r} must be " f"denied 403, got {result.status_code}: {result.body[:300]}" ) - assert MODEL_ACCESS_DENIED_MARKER in result.body, ( - f"403 body must be a model-access denial, got: {result.body[:300]}" - ) + assert ( + MODEL_ACCESS_DENIED_MARKER in result.body + ), f"403 body must be a model-access denial, got: {result.body[:300]}" def test_llm_only_key_forbidden_from_management_route_403( self, client: AccessControlClient, resources: ResourceManager @@ -65,9 +73,9 @@ class TestAccessControl: f"llm-only key calling a management route must be denied 403, got " f"{result.status_code}: {result.body[:300]}" ) - assert ROUTE_NOT_ALLOWED_MARKER in result.body, ( - f"403 body must be a route-permission denial, got: {result.body[:300]}" - ) + assert ( + ROUTE_NOT_ALLOWED_MARKER in result.body + ), f"403 body must be a route-permission denial, got: {result.body[:300]}" def test_unknown_model_returns_400( self, client: AccessControlClient, resources: ResourceManager @@ -80,4 +88,6 @@ class TestAccessControl: f"unknown model must be rejected 400 before forwarding, got " f"{result.status_code}: {result.body[:300]}" ) - assert _is_json(result.body), f"400 body must be valid JSON: {result.body[:300]}" + assert _is_json( + result.body + ), f"400 body must be valid JSON: {result.body[:300]}" diff --git a/tests/e2e/budgets/test_budget_crud_e2e.py b/tests/e2e/budgets/test_budget_crud_e2e.py index e697eca0051..1bf5c2c78b8 100644 --- a/tests/e2e/budgets/test_budget_crud_e2e.py +++ b/tests/e2e/budgets/test_budget_crud_e2e.py @@ -12,11 +12,23 @@ import pytest from budget_client import BudgetClient from lifecycle import ResourceManager -pytestmark = pytest.mark.e2e +pytestmark = [ + pytest.mark.e2e, + pytest.mark.e2e_coverage( + module="budgets", + endpoint="/budget/*", + provider="proxy", + params=["budget_crud", "budget_duration"], + ), +] -def test_budget_crud_roundtrip(client: BudgetClient, resources: ResourceManager) -> None: - budget_id = client.create_budget(max_budget=12.5, soft_budget=10.0, budget_duration="30d") +def test_budget_crud_roundtrip( + client: BudgetClient, resources: ResourceManager +) -> None: + budget_id = client.create_budget( + max_budget=12.5, soft_budget=10.0, budget_duration="30d" + ) resources.defer(lambda: client.delete_budget(budget_id)) rows = client.budget_info(budget_id) @@ -31,19 +43,23 @@ def test_budget_crud_roundtrip(client: BudgetClient, resources: ResourceManager) resources.defer(lambda: client.delete_key(key)) info = client.gateway.key_info(key) linked = info.litellm_budget_table - assert info.budget_id == budget_id or (linked is not None and linked.max_budget == 12.5), ( - f"key does not reflect attached budget: {info.budget_id}, {linked}" - ) + assert info.budget_id == budget_id or ( + linked is not None and linked.max_budget == 12.5 + ), f"key does not reflect attached budget: {info.budget_id}, {linked}" -def test_budget_delete_removes_it(client: BudgetClient, resources: ResourceManager) -> None: +def test_budget_delete_removes_it( + client: BudgetClient, resources: ResourceManager +) -> None: budget_id = client.create_budget(max_budget=1.0) resources.defer(lambda: client.delete_budget(budget_id)) client.delete_budget(budget_id) assert not client.budget_info(budget_id), "budget still present after delete" -def test_budget_duration_schedules_reset_on_key(client: BudgetClient, resources: ResourceManager) -> None: +def test_budget_duration_schedules_reset_on_key( + client: BudgetClient, resources: ResourceManager +) -> None: key = client.generate_key(max_budget=10.0, budget_duration="30d") resources.defer(lambda: client.delete_key(key)) @@ -57,4 +73,6 @@ def test_budget_duration_schedules_reset_on_key(client: BudgetClient, resources: # get current time -> assert budget from days_left - budget_duration == days_left reset_dt = datetime.fromisoformat(str(reset_at).replace("Z", "+00:00")) days_out = (reset_dt - datetime.now(timezone.utc)).total_seconds() / 86400 - assert 0 < days_out < 40, f"reset should be scheduled ahead, got {days_out:.1f}d out" + assert ( + 0 < days_out < 40 + ), f"reset should be scheduled ahead, got {days_out:.1f}d out" diff --git a/tests/e2e/budgets/test_budget_enforcement_e2e.py b/tests/e2e/budgets/test_budget_enforcement_e2e.py index d03288b637a..4c9525a3c18 100644 --- a/tests/e2e/budgets/test_budget_enforcement_e2e.py +++ b/tests/e2e/budgets/test_budget_enforcement_e2e.py @@ -21,7 +21,16 @@ from e2e_config import unique_marker from e2e_http import require_successful_call from lifecycle import run_case -pytestmark = pytest.mark.e2e +pytestmark = [ + pytest.mark.e2e, + pytest.mark.e2e_coverage( + module="budgets", + endpoint="/chat/completions", + provider="proxy", + params=["budget_enforcement"], + ), +] + def _assert_budget_blocks(client: BudgetClient, key: str, *, user: str = "") -> None: """Send paid calls until the entity's budget blocks one. Key/user/org/member @@ -142,7 +151,5 @@ def _case_id(case_cls: Type[_BudgetCase]) -> str: ], ids=_case_id, ) -def test_budget_enforcement( - client: BudgetClient, case_cls: Type[_BudgetCase] -) -> None: +def test_budget_enforcement(client: BudgetClient, case_cls: Type[_BudgetCase]) -> None: run_case(case_cls(client)) diff --git a/tests/e2e/budgets/test_budget_fallback_e2e.py b/tests/e2e/budgets/test_budget_fallback_e2e.py index 6114bfe6fc9..08f87ce8a07 100644 --- a/tests/e2e/budgets/test_budget_fallback_e2e.py +++ b/tests/e2e/budgets/test_budget_fallback_e2e.py @@ -13,7 +13,15 @@ from budget_client import BudgetClient, model_budget from e2e_config import unique_marker from lifecycle import ResourceManager -pytestmark = pytest.mark.e2e +pytestmark = [ + pytest.mark.e2e, + pytest.mark.e2e_coverage( + module="budgets", + endpoint="/v1/messages", + provider="proxy", + params=["budget_fallback"], + ), +] PRIMARY_MODEL = "claude-haiku-4-5" FALLBACK_MODEL = "gpt-5.5" @@ -46,15 +54,15 @@ def test_budget_fallback_reroutes_anthropic_messages_to_openai( if FALLBACK_MODEL in served_by: break time.sleep(1) - assert served_by is not None and FALLBACK_MODEL in served_by, ( - f"{PRIMARY_MODEL}'s budget_fallbacks never rerouted to {FALLBACK_MODEL}" - ) + assert ( + served_by is not None and FALLBACK_MODEL in served_by + ), f"{PRIMARY_MODEL}'s budget_fallbacks never rerouted to {FALLBACK_MODEL}" # The rerouted call must be recorded under the fallback model, not the # exhausted primary - proving spend tracking followed the reroute. rows = client.gateway.poll_logs_for_key( key, predicate=lambda rows: any(FALLBACK_MODEL in (r.model or "") for r in rows) ) - assert any(FALLBACK_MODEL in (r.model or "") for r in rows), ( - f"no spend log recorded against {FALLBACK_MODEL} after the reroute" - ) + assert any( + FALLBACK_MODEL in (r.model or "") for r in rows + ), f"no spend log recorded against {FALLBACK_MODEL} after the reroute" diff --git a/tests/e2e/budgets/test_budget_reset_e2e.py b/tests/e2e/budgets/test_budget_reset_e2e.py index dcf776db9a2..84dd3706f1a 100644 --- a/tests/e2e/budgets/test_budget_reset_e2e.py +++ b/tests/e2e/budgets/test_budget_reset_e2e.py @@ -17,7 +17,15 @@ from e2e_config import unique_marker from e2e_http import require_successful_call from lifecycle import ResourceManager -pytestmark = pytest.mark.e2e +pytestmark = [ + pytest.mark.e2e, + pytest.mark.e2e_coverage( + module="budgets", + endpoint="/chat/completions", + provider="proxy", + params=["budget_reset"], + ), +] def _call(client: BudgetClient, key: str): diff --git a/tests/e2e/budgets/test_model_max_budget_e2e.py b/tests/e2e/budgets/test_model_max_budget_e2e.py index 44e6a333ef0..28c155f349b 100644 --- a/tests/e2e/budgets/test_model_max_budget_e2e.py +++ b/tests/e2e/budgets/test_model_max_budget_e2e.py @@ -15,7 +15,15 @@ from e2e_config import unique_marker from e2e_http import require_successful_call from lifecycle import ResourceManager -pytestmark = pytest.mark.e2e +pytestmark = [ + pytest.mark.e2e, + pytest.mark.e2e_coverage( + module="budgets", + endpoint="/chat/completions", + provider="proxy", + params=["model_max_budget"], + ), +] CAPPED_MODEL = "claude-haiku-4-5" FREE_MODEL = "gemini-2.5-flash" @@ -51,7 +59,7 @@ def test_model_max_budget_isolates_per_model( # The other model shares the key but has its own (large) cap -> still works. other = _call(client, key, FREE_MODEL) - assert not is_budget_block(other), ( - f"{FREE_MODEL} was blocked by {CAPPED_MODEL}'s budget; per-model caps not isolated" - ) + assert not is_budget_block( + other + ), f"{FREE_MODEL} was blocked by {CAPPED_MODEL}'s budget; per-model caps not isolated" require_successful_call(other) diff --git a/tests/e2e/budgets/test_multi_window_budget_e2e.py b/tests/e2e/budgets/test_multi_window_budget_e2e.py index 553ad1ce701..fc4a151d731 100644 --- a/tests/e2e/budgets/test_multi_window_budget_e2e.py +++ b/tests/e2e/budgets/test_multi_window_budget_e2e.py @@ -18,7 +18,15 @@ from e2e_http import require_successful_call from lifecycle import ResourceManager from models import BudgetWindow -pytestmark = pytest.mark.e2e +pytestmark = [ + pytest.mark.e2e, + pytest.mark.e2e_coverage( + module="budgets", + endpoint="/chat/completions", + provider="proxy", + params=["multi_window_budget"], + ), +] WINDOW_SECONDS = 30 # the tight window; calls succeed again only after it elapses @@ -62,9 +70,11 @@ def test_short_window_blocks_then_resets( result = _call(client, key) if result.ok: elapsed = time.monotonic() - start - assert elapsed < WINDOW_SECONDS + 90, ( - f"reset took {elapsed:.0f}s - too long for a {WINDOW_SECONDS}s window" - ) + assert ( + elapsed < WINDOW_SECONDS + 90 + ), f"reset took {elapsed:.0f}s - too long for a {WINDOW_SECONDS}s window" return - assert is_budget_block(result), f"non-budget error during reset wait: {result.body[:200]}" + assert is_budget_block( + result + ), f"non-budget error during reset wait: {result.body[:200]}" pytest.fail(f"{WINDOW_SECONDS}s window never reset within 150s") diff --git a/tests/e2e/budgets/test_soft_budget_e2e.py b/tests/e2e/budgets/test_soft_budget_e2e.py index 407de7ae467..142077432a3 100644 --- a/tests/e2e/budgets/test_soft_budget_e2e.py +++ b/tests/e2e/budgets/test_soft_budget_e2e.py @@ -14,7 +14,15 @@ from e2e_config import unique_marker from e2e_http import require_successful_call from lifecycle import ResourceManager -pytestmark = pytest.mark.e2e +pytestmark = [ + pytest.mark.e2e, + pytest.mark.e2e_coverage( + module="budgets", + endpoint="/chat/completions", + provider="proxy", + params=["soft_budget"], + ), +] def test_soft_budget_does_not_block( @@ -32,4 +40,6 @@ def test_soft_budget_does_not_block( "soft_budget blocked a request; it must alert only, not block " f"(body={result.body[:200]})" ) - require_successful_call(result) # any other non-2xx (e.g. provider down) is a hard fail + require_successful_call( + result + ) # any other non-2xx (e.g. provider down) is a hard fail diff --git a/tests/e2e/budgets/test_spend_counter_reseed_e2e.py b/tests/e2e/budgets/test_spend_counter_reseed_e2e.py index a6860aeef43..5856475f2e3 100644 --- a/tests/e2e/budgets/test_spend_counter_reseed_e2e.py +++ b/tests/e2e/budgets/test_spend_counter_reseed_e2e.py @@ -33,7 +33,15 @@ if TYPE_CHECKING: import redis from redis.cluster import RedisCluster -pytestmark = pytest.mark.e2e +pytestmark = [ + pytest.mark.e2e, + pytest.mark.e2e_coverage( + module="budgets", + endpoint="/chat/completions", + provider="proxy", + params=["spend_counter_reseed"], + ), +] MODEL = "claude-haiku-4-5" ACCUMULATE_CALLS = 24 @@ -51,7 +59,9 @@ def _redis() -> "redis.Redis[str] | RedisCluster[str]": host = os.getenv("REDIS_HOST") if not host: - return redis.Redis(host="localhost", port=6380, decode_responses=True, socket_connect_timeout=2) + return redis.Redis( + host="localhost", port=6380, decode_responses=True, socket_connect_timeout=2 + ) from redis.cluster import RedisCluster @@ -64,7 +74,9 @@ def _redis() -> "redis.Redis[str] | RedisCluster[str]": ) -def _spend_counter(rds: "redis.Redis[str] | RedisCluster[str]", key: str) -> float | None: +def _spend_counter( + rds: "redis.Redis[str] | RedisCluster[str]", key: str +) -> float | None: """The shared spend counter for `key`, or None if it is cold. A cluster client can't run a keyspace SCAN that spans shards, so read the key directly - the stage gateway sets no cache namespace, so the key is the bare ``spend:key:{sha256(key)}``. diff --git a/tests/e2e/budgets/test_tag_budget_e2e.py b/tests/e2e/budgets/test_tag_budget_e2e.py index 7cec5bc96c1..62b8448c70c 100644 --- a/tests/e2e/budgets/test_tag_budget_e2e.py +++ b/tests/e2e/budgets/test_tag_budget_e2e.py @@ -15,7 +15,15 @@ from e2e_config import unique_marker from e2e_http import require_successful_call from lifecycle import ResourceManager -pytestmark = pytest.mark.e2e +pytestmark = [ + pytest.mark.e2e, + pytest.mark.e2e_coverage( + module="budgets", + endpoint="/chat/completions", + provider="proxy", + params=["tag_budget"], + ), +] TINY_BUDGET = 1e-6 @@ -53,7 +61,7 @@ def test_tag_budget_blocks_tagged_requests( # A request with an unbudgeted tag on the same key is unaffected. free_tag = f"e2e-free-tag-{unique_marker()}" other = _tagged_call(client, scoped_key, free_tag) - assert not is_budget_block(other), ( - f"unbudgeted tag {free_tag!r} was blocked by {budgeted_tag!r}'s budget" - ) + assert not is_budget_block( + other + ), f"unbudgeted tag {free_tag!r} was blocked by {budgeted_tag!r}'s budget" require_successful_call(other) diff --git a/tests/e2e/budgets/test_team_member_budget_e2e.py b/tests/e2e/budgets/test_team_member_budget_e2e.py index 301617bfdca..85bffad268c 100644 --- a/tests/e2e/budgets/test_team_member_budget_e2e.py +++ b/tests/e2e/budgets/test_team_member_budget_e2e.py @@ -24,7 +24,15 @@ from e2e_http import Success, require_successful_call from lifecycle import ResourceManager from models import ChatBody, ChatMessage -pytestmark = pytest.mark.e2e +pytestmark = [ + pytest.mark.e2e, + pytest.mark.e2e_coverage( + module="budgets", + endpoint="/chat/completions", + provider="proxy", + params=["team_member_budget"], + ), +] MODEL = "claude-haiku-4-5" TEAM_BUDGET = 100.0 @@ -49,7 +57,9 @@ def member(client: BudgetClient) -> Iterator[_Member]: resources = ResourceManager(client=client.gateway) try: marker = unique_marker() - team_id = client.create_team(alias=f"e2e-team-member-{marker}", max_budget=TEAM_BUDGET) + team_id = client.create_team( + alias=f"e2e-team-member-{marker}", max_budget=TEAM_BUDGET + ) resources.defer(lambda: client.delete_team(team_id)) user_id = client.create_user(max_budget=TEAM_BUDGET) resources.defer(lambda: client.delete_user(user_id)) @@ -79,8 +89,12 @@ def _send(client: BudgetClient, key: str) -> str | None: class TestTeamMemberBudget: - def test_member_spend_attributed_to_team_and_user(self, client: BudgetClient, member: _Member) -> None: - sent = frozenset(rid for rid in (_send(client, member.key) for _ in range(BURST)) if rid) + def test_member_spend_attributed_to_team_and_user( + self, client: BudgetClient, member: _Member + ) -> None: + sent = frozenset( + rid for rid in (_send(client, member.key) for _ in range(BURST)) if rid + ) assert sent, "no member call went through; cannot check attribution" rows = client.gateway.poll_logs_for_key( @@ -90,16 +104,20 @@ class TestTeamMemberBudget: assert logged, f"none of the member's {len(sent)} calls reached the spend logs" for row in logged: - assert row.team_id == member.team_id, ( - f"call {row.request_id} logged under team {row.team_id}, not the member's team {member.team_id}" - ) - assert row.user == member.user_id, ( - f"call {row.request_id} logged under user {row.user}, not member {member.user_id}" - ) + assert ( + row.team_id == member.team_id + ), f"call {row.request_id} logged under team {row.team_id}, not the member's team {member.team_id}" + assert ( + row.user == member.user_id + ), f"call {row.request_id} logged under user {row.user}, not member {member.user_id}" - def test_member_spend_over_budget_is_blocked(self, client: BudgetClient, member: _Member) -> None: + def test_member_spend_over_budget_is_blocked( + self, client: BudgetClient, member: _Member + ) -> None: for _ in range(40): - result = client.chat(member.key, MODEL, f"spend {unique_marker()}", max_tokens=16) + result = client.chat( + member.key, MODEL, f"spend {unique_marker()}", max_tokens=16 + ) if is_budget_block(result): return require_successful_call(result) diff --git a/tests/e2e/budgets/test_team_member_budget_reset_e2e.py b/tests/e2e/budgets/test_team_member_budget_reset_e2e.py index 2749f16a26e..4b0270af22d 100644 --- a/tests/e2e/budgets/test_team_member_budget_reset_e2e.py +++ b/tests/e2e/budgets/test_team_member_budget_reset_e2e.py @@ -8,32 +8,51 @@ from e2e_config import unique_marker from e2e_http import require_successful_call from lifecycle import ResourceManager -pytestmark = pytest.mark.e2e +pytestmark = [ + pytest.mark.e2e, + pytest.mark.e2e_coverage( + module="budgets", + endpoint="/chat/completions", + provider="proxy", + params=["team_member_budget_reset"], + ), +] + +MEMBER_BUDGET = 1.0 # default member budget is $50, we're testing with a smaller value -MEMBER_BUDGET = 1.0 # default member budget is $50, we're testing with a smaller value def _as_datetime(value: str) -> datetime: return datetime.fromisoformat(value.replace("Z", "+00:00")) -def test_team_member_budget_reset_keeps_advancing(client: BudgetClient, resources: ResourceManager) -> None: - team_id = client.create_team(alias=f"e2e-member-reset-{unique_marker()}", max_budget=100.0) +def test_team_member_budget_reset_keeps_advancing( + client: BudgetClient, resources: ResourceManager +) -> None: + team_id = client.create_team( + alias=f"e2e-member-reset-{unique_marker()}", max_budget=100.0 + ) resources.defer(lambda: client.delete_team(team_id)) user_id = client.create_user(max_budget=100.0) resources.defer(lambda: client.delete_user(user_id)) # add the member, then update them onto a short per-team budget window client.add_team_member(team_id, user_id, max_budget_in_team=MEMBER_BUDGET) - client.update_team_member(team_id, user_id, max_budget_in_team=MEMBER_BUDGET, budget_duration="30s") + client.update_team_member( + team_id, user_id, max_budget_in_team=MEMBER_BUDGET, budget_duration="30s" + ) scheduled = client.member_budget_reset_at(team_id, user_id) - assert scheduled, "updating the member with a budget_duration set no budget_reset_at" + assert ( + scheduled + ), "updating the member with a budget_duration set no budget_reset_at" first_reset = _as_datetime(scheduled) # the member can spend within the team while the window is live key = client.generate_key(team_id=team_id, user_id=user_id) resources.defer(lambda: client.delete_key(key)) - require_successful_call(client.chat(key, "claude-haiku-4-5", f"reset {unique_marker()}", max_tokens=16)) + require_successful_call( + client.chat(key, "claude-haiku-4-5", f"reset {unique_marker()}", max_tokens=16) + ) # once the window elapses the reset job must move budget_reset_at forward; a job # that skips the member's budget row (the #25109 regression) leaves it pinned at @@ -44,4 +63,6 @@ def test_team_member_budget_reset_keeps_advancing(client: BudgetClient, resource current = client.member_budget_reset_at(team_id, user_id) if current and _as_datetime(current) > first_reset: return - pytest.fail(f"member budget_reset_at never advanced past {first_reset.isoformat()} in 150s") + pytest.fail( + f"member budget_reset_at never advanced past {first_reset.isoformat()} in 150s" + ) diff --git a/tests/e2e/budgets/test_team_multi_window_budget_e2e.py b/tests/e2e/budgets/test_team_multi_window_budget_e2e.py index c58e74db965..7c7963e4026 100644 --- a/tests/e2e/budgets/test_team_multi_window_budget_e2e.py +++ b/tests/e2e/budgets/test_team_multi_window_budget_e2e.py @@ -23,16 +23,28 @@ from e2e_http import require_successful_call from lifecycle import ResourceManager from models import BudgetWindow -pytestmark = pytest.mark.e2e +pytestmark = [ + pytest.mark.e2e, + pytest.mark.e2e_coverage( + module="budgets", + endpoint="/chat/completions", + provider="proxy", + params=["team_multi_window_budget"], + ), +] WINDOW_SECONDS = 30 def _call(client: BudgetClient, key: str): - return client.chat(key, "claude-haiku-4-5", f"team-window {unique_marker()}", max_tokens=16) + return client.chat( + key, "claude-haiku-4-5", f"team-window {unique_marker()}", max_tokens=16 + ) -def test_team_short_window_blocks_then_resets(client: BudgetClient, resources: ResourceManager) -> None: +def test_team_short_window_blocks_then_resets( + client: BudgetClient, resources: ResourceManager +) -> None: team_id = client.create_team( alias=f"e2e-team-window-{unique_marker()}", budget_limits=[ @@ -66,7 +78,11 @@ def test_team_short_window_blocks_then_resets(client: BudgetClient, resources: R result = _call(client, key) if result.ok: elapsed = time.monotonic() - start - assert elapsed < WINDOW_SECONDS + 90, f"reset took {elapsed:.0f}s - too long for a {WINDOW_SECONDS}s window" + assert ( + elapsed < WINDOW_SECONDS + 90 + ), f"reset took {elapsed:.0f}s - too long for a {WINDOW_SECONDS}s window" return - assert is_budget_block(result), f"non-budget error during reset wait: {result.body[:200]}" + assert is_budget_block( + result + ), f"non-budget error during reset wait: {result.body[:200]}" pytest.fail(f"team {WINDOW_SECONDS}s window never reset within 150s") diff --git a/tests/e2e/conftest.py b/tests/e2e/conftest.py index 9ca5840df24..039155795e5 100644 --- a/tests/e2e/conftest.py +++ b/tests/e2e/conftest.py @@ -24,7 +24,6 @@ import requests from e2e_config import CONTROL_PLANE_BASE_URL, PROXY_BASE_URL from lifecycle import GatewayProvider, ResourceManager - _E2E_TEST_RAN = pytest.StashKey[bool]() @@ -35,7 +34,7 @@ def pytest_configure(config: pytest.Config) -> None: ) config.addinivalue_line( "markers", - "covers(cell_id, *, exercised_on=()): coverage-registry cell(s) this test covers", + "e2e_coverage(module, endpoint, provider, params): structured e2e coverage metadata", ) @@ -102,7 +101,7 @@ def pytest_sessionfinish(session: pytest.Session, exitstatus: int) -> None: reset_spend_logs() except Exception as exc: # noqa: BLE001 - cleanup is best-effort - print(f"spend-log cleanup skipped: {exc}") + print(f"spend-log cleanup skipped: {exc}") # noqa: T201 finally: if spend_dir in sys.path: sys.path.remove(spend_dir) diff --git a/tests/e2e/coverage_registry/README.md b/tests/e2e/coverage_registry/README.md index 863ce34694d..24753f75b54 100644 --- a/tests/e2e/coverage_registry/README.md +++ b/tests/e2e/coverage_registry/README.md @@ -1,81 +1,90 @@ -# e2e coverage registry +# e2e coverage metadata -This directory is the **denominator** for e2e test coverage: the set of behaviors we -want covered, one row per behavior, checked into the repo so coverage is a number we -can track instead of a guess. It implements the plan in the "E2E Coverage Tracking" -note; the naming grammar lives in `tests/e2e/CLAUDE.md`. +E2E coverage is declared directly on pytest tests. There are no coverage YAML +files. -## The model - -A **cell** is one customer-noticeable behavior a single e2e test can assert pass/fail -on, for example `llm.chat_completions.bedrock_converse.tool_use.stream.works`. Cells are -grouped `module > feature > test`, with LLM cells split into `Core LLMs` and -`Non-Core LLMs` for dashboarding. Each cell carries a tier (P0/P1/P2), a source, and a -`fail_before_fix` flag. - -The rows live in per-prefix YAML files (`llm_*.yaml`, `mgmt.yaml`, `mcp.yaml`, -`reliability.yaml`, `logging.yaml`, `guardrail.yaml`, `other.yaml`) and validate against -the discriminated union in `schema.py`, so an LLM row cannot carry a guardrail field and -vice versa. `llm` rows with `subject_endpoint` of `chat_completions`, `messages`, or -`responses` roll up to `Core LLMs`; all other LLM endpoints roll up to `Non-Core LLMs`. -LLM endpoint, route, and capability values are typed in `schema.py`, so new taxonomy -values require an explicit schema change. `logging` and `guardrail` are two id-prefixes -that roll up into the single `Logging & Guardrails` dashboard module. - -A test declares what it covers with a marker: +Each collected pytest under `tests/e2e` must have: ```python -@pytest.mark.covers("llm.chat_completions.openai.tool_use.stream.works") -def test_openai_streaming_tool_calls(self) -> None: - ... +pytestmark = [ + pytest.mark.e2e, + pytest.mark.e2e_coverage( + module="core_llms", + endpoint="/chat/completions", + provider="openai", + params=["tools", "streaming"], + ), +] ``` -## The number +Use a module-level `pytestmark` when every test in a file covers the same +surface. Use a per-test marker when a file mixes endpoints, providers, or params. -`collector.py` diffs the registry against those markers and reports coverage per module. -It is static: a collect-only pass reads the markers, so it runs no test and needs no live -proxy. Whether a covered cell currently passes or fails is a separate, live concern. +## Required fields -``` -cd tests/e2e && PYTHONPATH=. python -m coverage_registry.collector +- `module`: one of the known dashboard modules in `schema.py` +- `endpoint`: a known endpoint or surface, such as `/chat/completions`, + `/v1/messages`, `/v1/batches`, `/budget/*`, or `/spend/*` +- `provider`: a known provider/integration, such as `proxy`, `openai`, + `anthropic`, `vertex_ai`, `prometheus`, or `multiple` +- `params`: one or more explicit lowercase parameter/behavior names + +## How coverage is measured + +The collector runs pytest in collect-only mode, validates `e2e_coverage` markers, +and expands each marker into unique: + +```text +module x endpoint x provider x param ``` -Use `--format loki` after the e2e pytest run in the same Kubernetes job/pod to print -structured stdout lines for Loki: +coverage units. -``` -cd tests/e2e && PYTHONPATH=. python -m coverage_registry.collector --format loki --strict +The report shows: + +- unique coverage units by module +- collected pytest test count by module +- endpoint/provider/param breakdowns for Grafana tables +- missing or invalid metadata counts for CI + +This answers questions like: + +```text +How many /chat/completions rate-limit params do we exercise? +How many budget tests exist? +Which modules gained or lost endpoint coverage over time? ``` -This emits exactly one `COVERAGE_TOTAL` line and one `COVERAGE_MODULE` line per module -in `MODULE_ORDER`, in that order. Loki uses log-safe `module=` labels from -`LOKI_MODULE_LABELS` (`core_llms`, `management_ui`, etc.) so existing JSON and -Prometheus consumers keep their human-readable module names unchanged. +## Commands -The headline is overall coverage. The collector also lists markers that point at ids -not in the registry, so a typo or an unenumerated behavior surfaces instead of being -silently dropped. +Render the report: -Use strict mode in CI once existing draft markers are reconciled: - -``` -cd tests/e2e && PYTHONPATH=. python -m coverage_registry.collector --strict +```bash +cd tests/e2e +PYTHONPATH=. python -m coverage_registry.collector ``` -Strict mode exits non-zero on `@pytest.mark.covers(...)` ids that are not checked into -the registry. Add `--fail-on-collection-errors` when the job should also fail on pytest -collection errors. +Emit Loki lines after the e2e job: -## Status: this is a draft for review +```bash +cd tests/e2e +PYTHONPATH=. python -m coverage_registry.collector --format loki --strict +``` -The cells were enumerated from the codebase and the tiers are a first proposal. Known -things to settle before treating the set as final: +Validate every collected pytest has metadata: -- tiers are proposed, not signed off; 125 P0 is a lot to prove fail-before-fix, so P0 may - want tightening -- a few cells need a support check or a prune (for example `llm.embeddings.anthropic.*` - and `reliability.perf.throughput.under_slo`) -- auth is covered in two places (`other.auth.*` and the mgmt authz assertions); the - boundary needs a decision, and the auth cluster may deserve promotion to its own module -- the P2 "niche" cells each stand in for a large tail of integrations/providers by design, - so the denominator is deliberately P0-weighted rather than a full inventory +```bash +cd tests/e2e +PYTHONPATH=. python -m coverage_registry.check_coverage_sync +``` + +CI runs the sync check in `.github/workflows/test-code-quality.yml`. + +## Adding coverage + +1. Add the pytest. +2. Add `@pytest.mark.e2e_coverage(...)` or module-level `pytestmark`. +3. Pick the closest module, endpoint, provider, and params. +4. If the endpoint or provider is real but rejected, add it to `schema.py`. +5. Run `PYTHONPATH=. python -m coverage_registry.check_coverage_sync` from + `tests/e2e`. diff --git a/tests/e2e/coverage_registry/__init__.py b/tests/e2e/coverage_registry/__init__.py index 959b3327194..a683155e3eb 100644 --- a/tests/e2e/coverage_registry/__init__.py +++ b/tests/e2e/coverage_registry/__init__.py @@ -1,8 +1,5 @@ -"""The e2e coverage registry: the denominator for e2e test coverage. +"""Marker-based e2e coverage metadata. -`schema.py` defines one validated row per customer-noticeable behavior (a "cell"). -The `*.yaml` files hold the rows, one file per id-prefix. `registry.py` loads and -validates them; `collector.py` diffs the registry against the `@pytest.mark.covers` -markers on the live tests and reports coverage per module. See tests/e2e/CLAUDE.md -for the naming grammar. +`schema.py` validates `@pytest.mark.e2e_coverage(...)` fields and `collector.py` +turns collected pytest markers into module, endpoint, provider, and param counts. """ diff --git a/tests/e2e/coverage_registry/check_coverage_sync.py b/tests/e2e/coverage_registry/check_coverage_sync.py new file mode 100644 index 00000000000..d1cdc54a4e7 --- /dev/null +++ b/tests/e2e/coverage_registry/check_coverage_sync.py @@ -0,0 +1,80 @@ +"""Validate that every collected e2e pytest declares coverage metadata. + +Run from tests/e2e: + + PYTHONPATH=. python -m coverage_registry.check_coverage_sync + +This is the CI guardrail for coverage drift. A new pytest fails this check until +it declares module, endpoint, provider, and params via @pytest.mark.e2e_coverage. +""" + +from __future__ import annotations + +import sys +from argparse import ArgumentParser +from pathlib import Path + +from .collector import E2E_DIR, collect_coverage_markers, compute_coverage + +IGNORED_SUITE_DIRS = frozenset({".pytest_cache", "__pycache__"}) + + +def suite_dirs_missing_tests(e2e_dir: Path = E2E_DIR) -> tuple[Path, ...]: + """Return top-level suite dirs that contain no pytest tests.""" + return tuple( + sorted( + path + for path in e2e_dir.iterdir() + if path.is_dir() + and path.name not in IGNORED_SUITE_DIRS + and not any(path.rglob("test_*.py")) + ) + ) + + +def sync_errors(e2e_dir: Path = E2E_DIR) -> tuple[str, ...]: + """Return human-readable sync errors for CI output.""" + report = compute_coverage(collect_coverage_markers(e2e_dir)) + + errors: list[str] = [] + if report.collection_errors: + errors.append( + "Pytest collection errors while reading coverage markers:\n " + + "\n ".join(report.collection_errors) + ) + if report.invalid_markers: + errors.append( + "Invalid @pytest.mark.e2e_coverage metadata:\n " + + "\n ".join(report.invalid_markers) + ) + if report.unmarked_nodeids: + errors.append( + "Collected e2e tests missing @pytest.mark.e2e_coverage. Add module, " + "endpoint, provider, and params fields, for example:\n\n" + " @pytest.mark.e2e_coverage(\n" + ' module="core_llms",\n' + ' endpoint="/chat/completions",\n' + ' provider="openai",\n' + ' params=["tools"],\n' + " )\n\n" + "Unmapped tests:\n " + "\n ".join(report.unmarked_nodeids) + ) + return tuple(errors) + + +def main() -> int: + parser = ArgumentParser() + parser.parse_args() + + errors = sync_errors() + if errors: + print("E2E coverage metadata is out of sync:\n", file=sys.stderr) + print("\n\n".join(errors), file=sys.stderr) + return 1 + + print("E2E coverage metadata is in sync.") # noqa: T201 + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/tests/e2e/coverage_registry/collector.py b/tests/e2e/coverage_registry/collector.py index f6e59ca4a88..55907ac2a2f 100644 --- a/tests/e2e/coverage_registry/collector.py +++ b/tests/e2e/coverage_registry/collector.py @@ -1,11 +1,17 @@ -"""Diff the registry (denominator) against the @pytest.mark.covers markers on the -live tests (numerator) and report coverage per module. +"""Collect e2e coverage metadata from pytest markers. -Coverage here is static: it reads the markers via a collect-only pass, so it runs -no test and needs no live proxy. Whether a covered cell currently passes or fails -(covered_pass vs covered_fail) is a separate, live concern layered on top later. +Coverage is declared directly on tests: - cd tests/e2e && PYTHONPATH=. python -m coverage_registry.collector + @pytest.mark.e2e_coverage( + module="core_llms", + endpoint="/chat/completions", + provider="openai", + params=["tools"], + ) + +The collector runs pytest in collect-only mode, validates those markers, expands +them into endpoint x provider x parameter units, and renders Grafana/Loki-friendly +summaries. It does not execute live e2e tests. """ from __future__ import annotations @@ -17,42 +23,82 @@ import sys from argparse import ArgumentParser from dataclasses import dataclass from pathlib import Path +from typing import Mapping import pytest +from pydantic import ValidationError -from .registry import load_registry -from .schema import MODULE_ORDER, Cell, Tier, dashboard_module, loki_module_label +from .schema import ( + MODULE_DISPLAY_NAMES, + MODULE_ORDER, + CoverageModule, + CoveragePoint, + CoverageUnit, + units_for_point, +) E2E_DIR = Path(__file__).resolve().parent.parent -class _CoversSink: - """Pytest plugin: after collection, capture every cell id declared via - @pytest.mark.covers(...), plus any nodes that failed to import.""" +class _CoverageSink: + """Pytest plugin that records e2e coverage markers after collection.""" def __init__(self) -> None: - self.covered_ids: frozenset[str] = frozenset() + self.points_by_nodeid: dict[str, tuple[CoveragePoint, ...]] = {} + self.collected_nodeids: tuple[str, ...] = () + self.unmarked_nodeids: tuple[str, ...] = () + self.invalid_markers: tuple[str, ...] = () self.collection_errors: tuple[str, ...] = () def pytest_collection_finish(self, session: pytest.Session) -> None: - self.covered_ids = frozenset( - arg - for item in session.items - for marker in item.iter_markers(name="covers") - for arg in marker.args - if isinstance(arg, str) - ) + collected_nodeids: list[str] = [] + unmarked_nodeids: list[str] = [] + invalid_markers: list[str] = [] + points_by_nodeid: dict[str, tuple[CoveragePoint, ...]] = {} + + for item in session.items: + collected_nodeids.append(item.nodeid) + markers = tuple(item.iter_markers(name="e2e_coverage")) + if not markers: + unmarked_nodeids.append(item.nodeid) + continue + + points: list[CoveragePoint] = [] + for marker in markers: + if marker.args: + invalid_markers.append( + f"{item.nodeid}: e2e_coverage uses keyword fields only" + ) + continue + try: + points.append(CoveragePoint.model_validate(marker.kwargs)) + except ValidationError as exc: + invalid_markers.append(f"{item.nodeid}: {exc}") + if points: + points_by_nodeid[item.nodeid] = tuple(points) + + self.points_by_nodeid = points_by_nodeid + self.collected_nodeids = tuple(sorted(collected_nodeids)) + self.unmarked_nodeids = tuple(sorted(unmarked_nodeids)) + self.invalid_markers = tuple(sorted(invalid_markers)) def pytest_collectreport(self, report: pytest.CollectReport) -> None: if report.failed: self.collection_errors = (*self.collection_errors, report.nodeid) -def collect_covered_ids( - e2e_dir: Path = E2E_DIR, -) -> tuple[frozenset[str], tuple[str, ...]]: - """Return (covered cell ids, nodeids that failed to import).""" - sink = _CoversSink() +@dataclass(frozen=True, slots=True) +class CoverageMarkers: + points_by_nodeid: Mapping[str, tuple[CoveragePoint, ...]] + collected_nodeids: tuple[str, ...] + unmarked_nodeids: tuple[str, ...] + invalid_markers: tuple[str, ...] + collection_errors: tuple[str, ...] + + +def collect_coverage_markers(e2e_dir: Path = E2E_DIR) -> CoverageMarkers: + """Return validated coverage markers plus collection diagnostics.""" + sink = _CoverageSink() with contextlib.redirect_stdout(io.StringIO()): pytest.main( [ @@ -65,109 +111,204 @@ def collect_covered_ids( ], plugins=[sink], ) - return sink.covered_ids, sink.collection_errors + return CoverageMarkers( + points_by_nodeid=sink.points_by_nodeid, + collected_nodeids=sink.collected_nodeids, + unmarked_nodeids=sink.unmarked_nodeids, + invalid_markers=sink.invalid_markers, + collection_errors=sink.collection_errors, + ) @dataclass(frozen=True, slots=True) class ModuleCoverage: - module: str - total: int - covered: int - p0_total: int - p0_covered: int + module: CoverageModule + unit_count: int + test_count: int + + @property + def display_name(self) -> str: + return MODULE_DISPLAY_NAMES[self.module] @property def coverage_percent(self) -> float: - return _percent(self.covered, self.total) + return 100.0 if self.unit_count else 0.0 + + +@dataclass(frozen=True, slots=True) +class EndpointCoverage: + module: CoverageModule + endpoint: str + provider: str + param_count: int + test_count: int + + @property + def display_name(self) -> str: + return MODULE_DISPLAY_NAMES[self.module] + + @property + def coverage_percent(self) -> float: + return 100.0 if self.param_count else 0.0 @dataclass(frozen=True, slots=True) class CoverageReport: + units: tuple[CoverageUnit, ...] modules: tuple[ModuleCoverage, ...] - total: int - covered: int - p0_total: int - p0_covered: int - p0_gaps: tuple[str, ...] - orphan_markers: tuple[str, ...] + endpoints: tuple[EndpointCoverage, ...] + test_count: int + collected_test_count: int + unmarked_test_count: int + invalid_marker_count: int + unmarked_nodeids: tuple[str, ...] + invalid_markers: tuple[str, ...] collection_errors: tuple[str, ...] + @property + def total(self) -> int: + return len(self.units) + + @property + def covered(self) -> int: + return len(self.units) + @property def coverage_percent(self) -> float: - return _percent(self.covered, self.total) + return 100.0 if self.units else 0.0 -def _percent(covered: int, total: int) -> float: - return (100.0 * covered / total) if total else 0.0 +def _unique_units( + points_by_nodeid: Mapping[str, tuple[CoveragePoint, ...]], +) -> tuple[CoverageUnit, ...]: + units = { + unit.key: unit + for points in points_by_nodeid.values() + for point in points + for unit in units_for_point(point) + } + return tuple(units[key] for key in sorted(units)) def _module_coverage( - module: str, cells: tuple[Cell, ...], covered: frozenset[str] + module: CoverageModule, + units: tuple[CoverageUnit, ...], + points_by_nodeid: Mapping[str, tuple[CoveragePoint, ...]], ) -> ModuleCoverage: - in_module = tuple(c for c in cells if dashboard_module(c) == module) - p0 = tuple(c for c in in_module if c.tier is Tier.P0) + test_nodeids = frozenset( + nodeid + for nodeid, points in points_by_nodeid.items() + if any(point.module == module for point in points) + ) return ModuleCoverage( module=module, - total=len(in_module), - covered=sum(1 for c in in_module if c.id in covered), - p0_total=len(p0), - p0_covered=sum(1 for c in p0 if c.id in covered), + unit_count=sum(1 for unit in units if unit.module == module), + test_count=len(test_nodeids), ) +def _endpoint_coverage( + units: tuple[CoverageUnit, ...], + points_by_nodeid: Mapping[str, tuple[CoveragePoint, ...]], +) -> tuple[EndpointCoverage, ...]: + keys = sorted({(unit.module, unit.endpoint, unit.provider) for unit in units}) + rows: list[EndpointCoverage] = [] + for module, endpoint, provider in keys: + params = frozenset( + unit.param + for unit in units + if unit.module == module + and unit.endpoint == endpoint + and unit.provider == provider + ) + test_nodeids = frozenset( + nodeid + for nodeid, points in points_by_nodeid.items() + if any( + point.module == module + and point.endpoint == endpoint + and point.provider == provider + for point in points + ) + ) + rows.append( + EndpointCoverage( + module=module, + endpoint=endpoint, + provider=provider, + param_count=len(params), + test_count=len(test_nodeids), + ) + ) + return tuple(rows) + + def compute_coverage( - cells: tuple[Cell, ...], - covered: frozenset[str], - collection_errors: tuple[str, ...] = (), + markers: CoverageMarkers, ) -> CoverageReport: - p0_cells = tuple(c for c in cells if c.tier is Tier.P0) - registry_ids = frozenset(c.id for c in cells) + units = _unique_units(markers.points_by_nodeid) + mapped_nodeids = frozenset(markers.points_by_nodeid) return CoverageReport( - modules=tuple(_module_coverage(m, cells, covered) for m in MODULE_ORDER), - total=len(cells), - covered=sum(1 for c in cells if c.id in covered), - p0_total=len(p0_cells), - p0_covered=sum(1 for c in p0_cells if c.id in covered), - p0_gaps=tuple(sorted(c.id for c in p0_cells if c.id not in covered)), - orphan_markers=tuple(sorted(covered - registry_ids)), - collection_errors=collection_errors, + units=units, + modules=tuple( + _module_coverage(module, units, markers.points_by_nodeid) + for module in MODULE_ORDER + ), + endpoints=_endpoint_coverage(units, markers.points_by_nodeid), + test_count=len(mapped_nodeids), + collected_test_count=len(markers.collected_nodeids), + unmarked_test_count=len(markers.unmarked_nodeids), + invalid_marker_count=len(markers.invalid_markers), + unmarked_nodeids=markers.unmarked_nodeids, + invalid_markers=markers.invalid_markers, + collection_errors=markers.collection_errors, ) -def _row(label: str, covered: int, total: int) -> str: - frac = f"{covered}/{total}" - return f"{label:30}{frac:>12}{_percent(covered, total):>11.1f}%" +def _row(label: str, unit_count: int, test_count: int) -> str: + return f"{label:30}{unit_count:>12}{test_count:>9}" def render(report: CoverageReport) -> str: - rows = tuple(_row(m.module, m.covered, m.total) for m in report.modules) - lines = ( - f"{'MODULE':30}{'COVERED':>12}{'COVERAGE':>12}", - *rows, - "-" * 54, - _row("ALL", report.covered, report.total), - "", - f"Headline coverage: {report.covered}/{report.total} ({report.coverage_percent:.1f}%)", + rows = tuple( + _row(m.display_name, m.unit_count, m.test_count) for m in report.modules ) - orphans = ( + lines = ( + f"{'MODULE':30}{'UNITS':>12}{'TESTS':>9}", + *rows, + "-" * 51, + _row("ALL", report.total, report.test_count), + "", + f"Coverage metadata: {report.test_count}/{report.collected_test_count} tests mapped", + f"Unique endpoint x provider x parameter units: {report.total}", + f"Unmarked collected tests: {report.unmarked_test_count}", + f"Invalid coverage markers: {report.invalid_marker_count}", + ) + invalid = ( ( - f"\n{len(report.orphan_markers)} marker(s) point at ids not in the registry " - f"(reconcile: fix the marker or add the cell):\n " - + "\n ".join(report.orphan_markers), + f"\n{len(report.invalid_markers)} invalid marker(s):\n " + + "\n ".join(report.invalid_markers), ) - if report.orphan_markers + if report.invalid_markers + else () + ) + unmarked = ( + ( + f"\n{len(report.unmarked_nodeids)} collected test(s) missing " + "e2e_coverage:\n " + "\n ".join(report.unmarked_nodeids), + ) + if report.unmarked_nodeids else () ) warning = ( ( - f"\nWARNING: {len(report.collection_errors)} node(s) failed to import during " - f"collection, so coverage may undercount:\n " - + "\n ".join(report.collection_errors), + f"\nWARNING: {len(report.collection_errors)} node(s) failed to import " + "during collection:\n " + "\n ".join(report.collection_errors), ) if report.collection_errors else () ) - return "\n".join((*lines, *orphans, *warning)) + return "\n".join((*lines, *invalid, *unmarked, *warning)) def _report_dict(report: CoverageReport) -> dict[str, object]: @@ -175,18 +316,45 @@ def _report_dict(report: CoverageReport) -> dict[str, object]: "covered": report.covered, "total": report.total, "coverage_percent": report.coverage_percent, + "test_count": report.test_count, + "collected_test_count": report.collected_test_count, + "unmarked_test_count": report.unmarked_test_count, + "invalid_marker_count": report.invalid_marker_count, "modules": [ { "module": m.module, - "covered": m.covered, - "total": m.total, + "display_name": m.display_name, + "covered": m.unit_count, + "total": m.unit_count, "coverage_percent": m.coverage_percent, - "p0_covered": m.p0_covered, - "p0_total": m.p0_total, + "test_count": m.test_count, } for m in report.modules ], - "orphan_markers": list(report.orphan_markers), + "endpoints": [ + { + "module": e.module, + "display_name": e.display_name, + "endpoint": e.endpoint, + "provider": e.provider, + "covered": e.param_count, + "total": e.param_count, + "coverage_percent": e.coverage_percent, + "test_count": e.test_count, + } + for e in report.endpoints + ], + "units": [ + { + "module": unit.module, + "endpoint": unit.endpoint, + "provider": unit.provider, + "param": unit.param, + } + for unit in report.units + ], + "unmarked_nodeids": list(report.unmarked_nodeids), + "invalid_markers": list(report.invalid_markers), "collection_errors": list(report.collection_errors), } @@ -201,36 +369,35 @@ def _label_value(value: str) -> str: def render_prometheus(report: CoverageReport) -> str: lines = [ - "# HELP litellm_e2e_coverage_cells E2E coverage registry cells by module and state.", - "# TYPE litellm_e2e_coverage_cells gauge", + "# HELP litellm_e2e_coverage_units Unique endpoint x provider x parameter units covered by e2e tests.", + "# TYPE litellm_e2e_coverage_units gauge", ] for module in report.modules: label = _label_value(module.module) lines.append( - f'litellm_e2e_coverage_cells{{module="{label}",state="covered"}} {module.covered}' - ) - lines.append( - f'litellm_e2e_coverage_cells{{module="{label}",state="total"}} {module.total}' + f'litellm_e2e_coverage_units{{module="{label}"}} {module.unit_count}' ) lines.extend( [ - f'litellm_e2e_coverage_cells{{module="ALL",state="covered"}} {report.covered}', - f'litellm_e2e_coverage_cells{{module="ALL",state="total"}} {report.total}', - "# HELP litellm_e2e_coverage_percent E2E coverage percent by module.", - "# TYPE litellm_e2e_coverage_percent gauge", + f'litellm_e2e_coverage_units{{module="ALL"}} {report.total}', + "# HELP litellm_e2e_coverage_tests E2E pytest test count with coverage metadata.", + "# TYPE litellm_e2e_coverage_tests gauge", ] ) for module in report.modules: label = _label_value(module.module) lines.append( - f'litellm_e2e_coverage_percent{{module="{label}"}} {module.coverage_percent:.6f}' + f'litellm_e2e_coverage_tests{{module="{label}"}} {module.test_count}' ) lines.extend( [ - f'litellm_e2e_coverage_percent{{module="ALL"}} {report.coverage_percent:.6f}', - "# HELP litellm_e2e_coverage_orphan_markers Coverage markers not found in the registry.", - "# TYPE litellm_e2e_coverage_orphan_markers gauge", - f"litellm_e2e_coverage_orphan_markers {len(report.orphan_markers)}", + f'litellm_e2e_coverage_tests{{module="ALL"}} {report.test_count}', + "# HELP litellm_e2e_coverage_unmarked_tests E2E collected tests without e2e_coverage metadata.", + "# TYPE litellm_e2e_coverage_unmarked_tests gauge", + f"litellm_e2e_coverage_unmarked_tests {report.unmarked_test_count}", + "# HELP litellm_e2e_coverage_invalid_markers E2E tests with invalid e2e_coverage metadata.", + "# TYPE litellm_e2e_coverage_invalid_markers gauge", + f"litellm_e2e_coverage_invalid_markers {report.invalid_marker_count}", "# HELP litellm_e2e_coverage_collection_errors Pytest nodes that failed during collection.", "# TYPE litellm_e2e_coverage_collection_errors gauge", f"litellm_e2e_coverage_collection_errors {len(report.collection_errors)}", @@ -243,14 +410,17 @@ def render_loki(report: CoverageReport) -> str: lines = [ ( f"COVERAGE_TOTAL percent={report.coverage_percent:.1f} " - f"covered={report.covered} total={report.total}" + f"covered={report.covered} total={report.total} tests={report.test_count} " + f"unmarked_tests={report.unmarked_test_count} " + f"invalid_markers={report.invalid_marker_count}" ) ] lines.extend( ( - f"COVERAGE_MODULE module={loki_module_label(module.module)} " + f"COVERAGE_MODULE module={module.module} " f"percent={module.coverage_percent:.1f} " - f"covered={module.covered} total={module.total}" + f"covered={module.unit_count} total={module.unit_count} " + f"tests={module.test_count}" ) for module in report.modules ) @@ -268,7 +438,7 @@ def main() -> int: parser.add_argument( "--strict", action="store_true", - help="Exit non-zero if markers outside the registry are found.", + help="Exit non-zero if tests are missing or have invalid coverage metadata.", ) parser.add_argument( "--fail-on-collection-errors", @@ -276,9 +446,7 @@ def main() -> int: help="Exit non-zero if pytest collection errors are found.", ) args = parser.parse_args() - cells = load_registry() - covered, errors = collect_covered_ids() - report = compute_coverage(cells, covered, errors) + report = compute_coverage(collect_coverage_markers()) output = { "text": render, "json": render_json, @@ -286,7 +454,7 @@ def main() -> int: "loki": render_loki, }[args.format](report) print(output) # noqa: T201 # CLI entrypoint output - if args.strict and report.orphan_markers: + if args.strict and (report.unmarked_nodeids or report.invalid_markers): return 1 if args.fail_on_collection_errors and report.collection_errors: return 1 diff --git a/tests/e2e/coverage_registry/guardrail.yaml b/tests/e2e/coverage_registry/guardrail.yaml deleted file mode 100644 index 792cbaaff7c..00000000000 --- a/tests/e2e/coverage_registry/guardrail.yaml +++ /dev/null @@ -1,29 +0,0 @@ -# Guardrail enforcement (behavior features). Grounded in litellm/proxy/guardrails/guardrail_hooks/. -# Rolls up into the "Logging & Guardrails" dashboard module together with logging.* -- {id: guardrail.presidio.pre_call.masks, module: guardrail, tier: P0, hook_point: pre_call, assertions: [masks], exercised_on: [chat_completions, messages], source: "guardrail_hooks/presidio.py", rationale: "PII masking pre-call; data-leak blast radius"} -- {id: guardrail.presidio.post_call.masks, module: guardrail, tier: P0, hook_point: post_call, assertions: [masks], exercised_on: [chat_completions, messages], source: "guardrail_hooks/presidio.py", rationale: "Mask PII in model output"} -- {id: guardrail.presidio.logging_only.masks, module: guardrail, tier: P0, hook_point: logging_only, assertions: [masks], exercised_on: [chat_completions, messages], source: "guardrail_hooks/presidio.py", rationale: "Redact in logs without blocking"} -- {id: guardrail.bedrock.pre_call.blocks, module: guardrail, tier: P0, hook_point: pre_call, assertions: [blocks], exercised_on: [chat_completions], source: "guardrail_hooks/bedrock_guardrails.py", rationale: "AWS content guardrail blocks harmful input"} -- {id: guardrail.bedrock.during.blocks, module: guardrail, tier: P0, hook_point: during, assertions: [blocks], exercised_on: [chat_completions], source: "guardrail_hooks/bedrock_guardrails.py", rationale: "During-call moderation for streaming"} -- {id: guardrail.bedrock.post_call.blocks, module: guardrail, tier: P0, hook_point: post_call, assertions: [blocks], exercised_on: [chat_completions], source: "guardrail_hooks/bedrock_guardrails.py", rationale: "Block harmful output"} -- {id: guardrail.lakera.pre_call.blocks, module: guardrail, tier: P0, hook_point: pre_call, assertions: [blocks], exercised_on: [chat_completions, messages], source: "guardrail_hooks/lakera_ai_v2.py", rationale: "Prompt-injection block pre-execution"} -- {id: guardrail.lakera.post_call.blocks, module: guardrail, tier: P0, hook_point: post_call, assertions: [blocks], exercised_on: [chat_completions], source: "guardrail_hooks/lakera_ai_v2.py", rationale: "Post-call injection on multi-turn chains"} -- {id: guardrail.openai_moderations.pre_call.blocks, module: guardrail, tier: P0, hook_point: pre_call, assertions: [blocks], exercised_on: [chat_completions, messages], source: "guardrail_hooks/openai/moderations.py", rationale: "Content policy for regulated industries"} -- {id: guardrail.aim.pre_call.blocks, module: guardrail, tier: P1, hook_point: pre_call, assertions: [blocks], exercised_on: [chat_completions, messages], source: "guardrail_hooks/aim/aim.py", rationale: "Security guardrail malicious-input"} -- {id: guardrail.aim.post_call.blocks, module: guardrail, tier: P1, hook_point: post_call, assertions: [blocks], exercised_on: [chat_completions], source: "guardrail_hooks/aim/aim.py", rationale: "Output security check"} -- {id: guardrail.ibm_guardrails.pre_call.blocks, module: guardrail, tier: P1, hook_point: pre_call, assertions: [blocks], exercised_on: [chat_completions], source: "guardrail_hooks/ibm_guardrails/ibm_detector.py", rationale: "Enterprise multi-policy"} -- {id: guardrail.ibm_guardrails.post_call.blocks, module: guardrail, tier: P1, hook_point: post_call, assertions: [blocks], exercised_on: [chat_completions], source: "guardrail_hooks/ibm_guardrails/ibm_detector.py", rationale: "Output policy validation"} -- {id: guardrail.semantic_guard.pre_call.blocks, module: guardrail, tier: P1, hook_point: pre_call, assertions: [blocks], exercised_on: [chat_completions], source: "guardrail_hooks/semantic_guard", rationale: "Semantic policy compliance"} -- {id: guardrail.block_code_execution.pre_call.blocks, module: guardrail, tier: P1, hook_point: pre_call, assertions: [blocks], exercised_on: [chat_completions], source: "guardrail_hooks/block_code_execution", rationale: "Code-injection prevention"} -- {id: guardrail.tool_permission.pre_call.allows, module: guardrail, tier: P1, hook_point: pre_call, assertions: [allows], exercised_on: [chat_completions], source: "guardrail_hooks/tool_permission.py", rationale: "Grant allowed tools"} -- {id: guardrail.tool_permission.pre_call.blocks, module: guardrail, tier: P1, hook_point: pre_call, assertions: [blocks], exercised_on: [chat_completions], source: "guardrail_hooks/tool_permission.py", rationale: "Block unauthorized tools"} -- {id: guardrail.microsoft_purview.pre_call.blocks, module: guardrail, tier: P1, hook_point: pre_call, assertions: [blocks], exercised_on: [chat_completions], source: "guardrail_hooks/microsoft_purview/purview_dlp.py", rationale: "DLP sensitive-data disclosure"} -- {id: guardrail.headroom.pre_call.blocks, module: guardrail, tier: P1, hook_point: pre_call, assertions: [blocks], exercised_on: [chat_completions], source: "guardrail_hooks/headroom/headroom.py", rationale: "Anomaly detection threshold"} -- {id: guardrail.generic_guardrail_api.pre_call.blocks, module: guardrail, tier: P1, hook_point: pre_call, assertions: [blocks], exercised_on: [chat_completions], source: "guardrail_hooks/generic_guardrail_api/generic_guardrail_api.py", rationale: "Vendor-agnostic custom API"} -- {id: guardrail.pangea.pre_call.blocks, module: guardrail, tier: P1, hook_point: pre_call, assertions: [blocks], exercised_on: [chat_completions], source: "guardrail_hooks/pangea/pangea.py", rationale: "API security + DLP"} -- {id: guardrail.niche_providers.pre_call.blocks, module: guardrail, tier: P2, hook_point: pre_call, assertions: [blocks], exercised_on: [chat_completions], source: grammar, rationale: "SMOKE cohort: lasso/hiddenlayer/model_armor/qualifire/guardrails_ai/cato/cisco/akto/prompt_security/promptguard/zscaler/vigil/etc"} -- {id: guardrail.niche_providers.post_call.blocks, module: guardrail, tier: P2, hook_point: post_call, assertions: [blocks], exercised_on: [chat_completions], source: grammar, rationale: "SMOKE niche output filtering"} -- {id: guardrail.niche_providers.pre_call.allows, module: guardrail, tier: P2, hook_point: pre_call, assertions: [allows], exercised_on: [chat_completions], source: grammar, rationale: "SMOKE niche allow-path passthrough"} -- {id: guardrail.tool_policy.pre_call.blocks, module: guardrail, tier: P2, hook_point: pre_call, assertions: [blocks], exercised_on: [chat_completions], source: "guardrail_hooks/tool_policy/tool_policy_guardrail.py", rationale: "Tool-use policy enforcement"} -- {id: guardrail.mcp_security.pre_call.blocks, module: guardrail, tier: P2, hook_point: pre_call, assertions: [blocks], exercised_on: [mcp_operations], source: "guardrail_hooks/mcp_security", rationale: "MCP protocol security"} -- {id: guardrail.llm_as_a_judge.pre_call.blocks, module: guardrail, tier: P2, hook_point: pre_call, assertions: [blocks], exercised_on: [chat_completions], source: "guardrail_hooks/llm_as_a_judge", rationale: "LLM-based judgment guardrail"} diff --git a/tests/e2e/coverage_registry/llm_conversational.yaml b/tests/e2e/coverage_registry/llm_conversational.yaml deleted file mode 100644 index 7360aacb916..00000000000 --- a/tests/e2e/coverage_registry/llm_conversational.yaml +++ /dev/null @@ -1,54 +0,0 @@ -# LLM conversational endpoints (chat_completions, messages, responses). Grounded in proxy handlers + model_prices json. -- {id: llm.chat_completions.openai.basic.nonstream.works, module: llm, tier: P0, subject_endpoint: chat_completions, route: openai, capability: basic, streaming: nonstream, assertions: [works], source: "proxy_server.py:8455", rationale: "Core endpoint/route/capability"} -- {id: llm.chat_completions.openai.basic.stream.works, module: llm, tier: P0, subject_endpoint: chat_completions, route: openai, capability: basic, streaming: stream, assertions: [works], source: "proxy_server.py:8455", rationale: "Core streaming"} -- {id: llm.chat_completions.openai.basic.nonstream.cost_logged, module: llm, tier: P0, subject_endpoint: chat_completions, route: openai, capability: basic, streaming: nonstream, assertions: [works, cost_logged], source: "proxy_server.py:8455", rationale: "Cost logging regression catch"} -- {id: llm.chat_completions.openai.tool_use.nonstream.works, module: llm, tier: P0, subject_endpoint: chat_completions, route: openai, capability: tool_use, streaming: nonstream, assertions: [works], source: "model_prices json", rationale: "OpenAI function_calling; high usage"} -- {id: llm.chat_completions.openai.tool_use.stream.works, module: llm, tier: P0, subject_endpoint: chat_completions, route: openai, capability: tool_use, streaming: stream, assertions: [works], source: "model_prices json", rationale: "Tool calls over streaming"} -- {id: llm.chat_completions.openai.vision.nonstream.works, module: llm, tier: P0, subject_endpoint: chat_completions, route: openai, capability: vision, streaming: nonstream, assertions: [works], source: "model_prices json", rationale: "gpt-4o vision; high usage"} -- {id: llm.chat_completions.openai.prompt_cache_5m.nonstream.works, module: llm, tier: P0, subject_endpoint: chat_completions, route: openai, capability: prompt_cache_5m, streaming: nonstream, assertions: [works], source: "model_prices json", rationale: "Prompt caching cost optimization"} -- {id: llm.chat_completions.openai.service_tier.nonstream.works, module: llm, tier: P1, subject_endpoint: chat_completions, route: openai, capability: service_tier, streaming: nonstream, assertions: [works], source: "OpenAI service_tier param", rationale: "OpenAI scale-tier request option is forwarded and echoed"} -- {id: llm.chat_completions.openai.thinking.nonstream.works, module: llm, tier: P1, subject_endpoint: chat_completions, route: openai, capability: thinking, streaming: nonstream, assertions: [works], source: "model_prices json", rationale: "o-series reasoning; emerging"} -- {id: llm.chat_completions.openai.structured_output.nonstream.works, module: llm, tier: P1, subject_endpoint: chat_completions, route: openai, capability: structured_output, streaming: nonstream, assertions: [works], source: "model_prices json", rationale: "response_schema extraction"} -- {id: llm.chat_completions.anthropic.basic.nonstream.works, module: llm, tier: P0, subject_endpoint: chat_completions, route: anthropic, capability: basic, streaming: nonstream, assertions: [works], source: "proxy_server.py:8455", rationale: "P0 route translated to Anthropic"} -- {id: llm.chat_completions.anthropic.basic.stream.works, module: llm, tier: P0, subject_endpoint: chat_completions, route: anthropic, capability: basic, streaming: stream, assertions: [works], source: "proxy_server.py:8455", rationale: "Streaming translation"} -- {id: llm.chat_completions.anthropic.tool_use.nonstream.works, module: llm, tier: P0, subject_endpoint: chat_completions, route: anthropic, capability: tool_use, streaming: nonstream, assertions: [works], source: "model_prices json", rationale: "Claude tool_use; high usage"} -- {id: llm.chat_completions.anthropic.tool_use.stream.works, module: llm, tier: P0, subject_endpoint: chat_completions, route: anthropic, capability: tool_use, streaming: stream, assertions: [works], source: "model_prices json", rationale: "Streaming tool calls"} -- {id: llm.chat_completions.anthropic.vision.nonstream.works, module: llm, tier: P0, subject_endpoint: chat_completions, route: anthropic, capability: vision, streaming: nonstream, assertions: [works], source: "model_prices json", rationale: "Claude vision; high usage"} -- {id: llm.chat_completions.anthropic.prompt_cache_5m.nonstream.works, module: llm, tier: P0, subject_endpoint: chat_completions, route: anthropic, capability: prompt_cache_5m, streaming: nonstream, assertions: [works], source: "model_prices json", rationale: "Claude prompt caching"} -- {id: llm.chat_completions.anthropic.thinking.nonstream.works, module: llm, tier: P1, subject_endpoint: chat_completions, route: anthropic, capability: thinking, streaming: nonstream, assertions: [works], source: "model_prices json", rationale: "Claude extended thinking"} -- {id: llm.chat_completions.anthropic.structured_output.nonstream.works, module: llm, tier: P1, subject_endpoint: chat_completions, route: anthropic, capability: structured_output, streaming: nonstream, assertions: [works], source: "model_prices json", rationale: "Claude response_schema"} -- {id: llm.chat_completions.bedrock_converse.basic.nonstream.works, module: llm, tier: P0, subject_endpoint: chat_completions, route: bedrock_converse, capability: basic, streaming: nonstream, assertions: [works], source: "proxy_server.py:8455", rationale: "P0 route; Bedrock Converse unified"} -- {id: llm.chat_completions.bedrock_converse.basic.stream.works, module: llm, tier: P0, subject_endpoint: chat_completions, route: bedrock_converse, capability: basic, streaming: stream, assertions: [works], source: "proxy_server.py:8455", rationale: "Streaming over Converse"} -- {id: llm.chat_completions.bedrock_converse.tool_use.nonstream.works, module: llm, tier: P0, subject_endpoint: chat_completions, route: bedrock_converse, capability: tool_use, streaming: nonstream, assertions: [works], source: "model_prices json", rationale: "Converse function_calling; AWS adoption"} -- {id: llm.chat_completions.bedrock_converse.vision.nonstream.works, module: llm, tier: P0, subject_endpoint: chat_completions, route: bedrock_converse, capability: vision, streaming: nonstream, assertions: [works], source: "model_prices json", rationale: "Bedrock vision (Anthropic/Nova)"} -- {id: llm.chat_completions.bedrock_converse.prompt_cache_5m.nonstream.works, module: llm, tier: P1, subject_endpoint: chat_completions, route: bedrock_converse, capability: prompt_cache_5m, streaming: nonstream, assertions: [works], source: "model_prices json", rationale: "Anthropic-on-Bedrock caching"} -- {id: llm.chat_completions.bedrock_converse.thinking.nonstream.works, module: llm, tier: P1, subject_endpoint: chat_completions, route: bedrock_converse, capability: thinking, streaming: nonstream, assertions: [works], source: "model_prices json", rationale: "Anthropic thinking on Bedrock"} -- {id: llm.chat_completions.vertex.basic.nonstream.works, module: llm, tier: P0, subject_endpoint: chat_completions, route: vertex, capability: basic, streaming: nonstream, assertions: [works], source: "proxy_server.py:8455", rationale: "P0 route; Vertex AI"} -- {id: llm.chat_completions.vertex.basic.stream.works, module: llm, tier: P0, subject_endpoint: chat_completions, route: vertex, capability: basic, streaming: stream, assertions: [works], source: "proxy_server.py:8455", rationale: "Streaming over Vertex"} -- {id: llm.chat_completions.vertex.tool_use.nonstream.works, module: llm, tier: P0, subject_endpoint: chat_completions, route: vertex, capability: tool_use, streaming: nonstream, assertions: [works], source: "model_prices json", rationale: "Vertex Gemini function_calling"} -- {id: llm.chat_completions.vertex.vision.nonstream.works, module: llm, tier: P0, subject_endpoint: chat_completions, route: vertex, capability: vision, streaming: nonstream, assertions: [works], source: "model_prices json", rationale: "Gemini vision"} -- {id: llm.chat_completions.vertex.prompt_cache_5m.nonstream.works, module: llm, tier: P1, subject_endpoint: chat_completions, route: vertex, capability: prompt_cache_5m, streaming: nonstream, assertions: [works], source: "model_prices json", rationale: "Vertex Gemini prompt caching"} -- {id: llm.chat_completions.azure_openai.basic.nonstream.works, module: llm, tier: P0, subject_endpoint: chat_completions, route: azure_openai, capability: basic, streaming: nonstream, assertions: [works], source: "proxy_server.py:8455", rationale: "P0 route; Azure OpenAI deployments"} -- {id: llm.chat_completions.azure_openai.tool_use.nonstream.works, module: llm, tier: P0, subject_endpoint: chat_completions, route: azure_openai, capability: tool_use, streaming: nonstream, assertions: [works], source: "model_prices json", rationale: "Azure OpenAI function_calling"} -- {id: llm.chat_completions.azure_foundry.basic.nonstream.works, module: llm, tier: P1, subject_endpoint: chat_completions, route: azure_foundry, capability: basic, streaming: nonstream, assertions: [works], source: "proxy_server.py:8455", rationale: "Azure Foundry (azure_ai); newer, smoke"} -- {id: llm.messages.anthropic.basic.nonstream.works, module: llm, tier: P0, subject_endpoint: messages, route: anthropic, capability: basic, streaming: nonstream, assertions: [works], source: "anthropic_endpoints/endpoints.py:64", rationale: "Core endpoint; Anthropic Messages native"} -- {id: llm.messages.anthropic.basic.stream.works, module: llm, tier: P0, subject_endpoint: messages, route: anthropic, capability: basic, streaming: stream, assertions: [works], source: "anthropic_endpoints/endpoints.py:64", rationale: "Streaming Messages API"} -- {id: llm.messages.anthropic.basic.nonstream.cost_logged, module: llm, tier: P0, subject_endpoint: messages, route: anthropic, capability: basic, streaming: nonstream, assertions: [works, cost_logged], source: "anthropic_endpoints/endpoints.py:64", rationale: "Cost logged on passthrough"} -- {id: llm.messages.anthropic.tool_use.nonstream.works, module: llm, tier: P0, subject_endpoint: messages, route: anthropic, capability: tool_use, streaming: nonstream, assertions: [works], source: "model_prices json", rationale: "Tool calls via Messages API"} -- {id: llm.messages.anthropic.tool_use.stream.works, module: llm, tier: P0, subject_endpoint: messages, route: anthropic, capability: tool_use, streaming: stream, assertions: [works], source: "model_prices json", rationale: "Streaming tool calls"} -- {id: llm.messages.anthropic.vision.nonstream.works, module: llm, tier: P0, subject_endpoint: messages, route: anthropic, capability: vision, streaming: nonstream, assertions: [works], source: "model_prices json", rationale: "Vision via Messages API"} -- {id: llm.messages.anthropic.prompt_cache_5m.nonstream.works, module: llm, tier: P0, subject_endpoint: messages, route: anthropic, capability: prompt_cache_5m, streaming: nonstream, assertions: [works], source: "model_prices json", rationale: "Prompt caching via Messages API"} -- {id: llm.messages.anthropic.thinking.nonstream.works, module: llm, tier: P1, subject_endpoint: messages, route: anthropic, capability: thinking, streaming: nonstream, assertions: [works], source: "model_prices json", rationale: "Extended thinking via Messages API"} -- {id: llm.responses.openai.basic.nonstream.works, module: llm, tier: P0, subject_endpoint: responses, route: openai, capability: basic, streaming: nonstream, assertions: [works], source: "response_api_endpoints/endpoints.py:26", rationale: "Core endpoint; OpenAI Responses native"} -- {id: llm.responses.openai.basic.stream.works, module: llm, tier: P0, subject_endpoint: responses, route: openai, capability: basic, streaming: stream, assertions: [works], source: "response_api_endpoints/endpoints.py:26", rationale: "Streaming via /v1/responses"} -- {id: llm.responses.openai.basic.nonstream.cost_logged, module: llm, tier: P0, subject_endpoint: responses, route: openai, capability: basic, streaming: nonstream, assertions: [works, cost_logged], source: "response_api_endpoints/endpoints.py:26", rationale: "Cost logged on responses"} -- {id: llm.responses.openai.tool_use.nonstream.works, module: llm, tier: P0, subject_endpoint: responses, route: openai, capability: tool_use, streaming: nonstream, assertions: [works], source: "model_prices json", rationale: "Tool calls via Responses API"} -- {id: llm.responses.openai.vision.nonstream.works, module: llm, tier: P0, subject_endpoint: responses, route: openai, capability: vision, streaming: nonstream, assertions: [works], source: "model_prices json", rationale: "Vision via Responses API"} -- {id: llm.responses.anthropic.basic.nonstream.works, module: llm, tier: P1, subject_endpoint: responses, route: anthropic, capability: basic, streaming: nonstream, assertions: [works], source: "response_api_endpoints/endpoints.py:26", rationale: "Responses w/ Anthropic translation (smoke)"} -- {id: llm.responses.anthropic.tool_use.nonstream.works, module: llm, tier: P1, subject_endpoint: responses, route: anthropic, capability: tool_use, streaming: nonstream, assertions: [works], source: "model_prices json", rationale: "Responses tool calls w/ Anthropic"} -- {id: llm.responses.bedrock_converse.basic.nonstream.works, module: llm, tier: P1, subject_endpoint: responses, route: bedrock_converse, capability: basic, streaming: nonstream, assertions: [works], source: "response_api_endpoints/endpoints.py:26", rationale: "Responses w/ Bedrock Converse (smoke)"} -- {id: llm.responses.bedrock_converse.tool_use.nonstream.works, module: llm, tier: P1, subject_endpoint: responses, route: bedrock_converse, capability: tool_use, streaming: nonstream, assertions: [works], source: "model_prices json", rationale: "Responses tool calls w/ Converse"} -- {id: llm.responses.vertex.basic.nonstream.works, module: llm, tier: P1, subject_endpoint: responses, route: vertex, capability: basic, streaming: nonstream, assertions: [works], source: "response_api_endpoints/endpoints.py:26", rationale: "Responses w/ Vertex (smoke)"} -- {id: llm.responses.vertex.tool_use.nonstream.works, module: llm, tier: P1, subject_endpoint: responses, route: vertex, capability: tool_use, streaming: nonstream, assertions: [works], source: "model_prices json", rationale: "Responses tool calls w/ Vertex"} -- {id: llm.responses.azure_openai.basic.nonstream.works, module: llm, tier: P1, subject_endpoint: responses, route: azure_openai, capability: basic, streaming: nonstream, assertions: [works], source: "response_api_endpoints/endpoints.py:26", rationale: "Responses w/ Azure OpenAI (smoke)"} -- {id: llm.responses.azure_openai.tool_use.nonstream.works, module: llm, tier: P1, subject_endpoint: responses, route: azure_openai, capability: tool_use, streaming: nonstream, assertions: [works], source: "model_prices json", rationale: "Responses tool calls w/ Azure OpenAI"} diff --git a/tests/e2e/coverage_registry/llm_nonconversational.yaml b/tests/e2e/coverage_registry/llm_nonconversational.yaml deleted file mode 100644 index b01b219476d..00000000000 --- a/tests/e2e/coverage_registry/llm_nonconversational.yaml +++ /dev/null @@ -1,45 +0,0 @@ -# LLM non-conversational endpoints. Grounded in litellm/proxy endpoints + llms/ handlers. -- {id: llm.embeddings.openai.basic.nonstream.works, module: llm, tier: P0, subject_endpoint: embeddings, route: openai, capability: basic, streaming: nonstream, assertions: [works], source: "test_embeddings_endpoint_e2e.py:23", rationale: "Core endpoint, live vector response"} -- {id: llm.embeddings.openai.basic.nonstream.cost_logged, module: llm, tier: P0, subject_endpoint: embeddings, route: openai, capability: basic, streaming: nonstream, assertions: [cost_logged], source: "SPEND_TRACKING_COVERAGE_MATRIX.md:34", rationale: "Cost tracking on embeddings"} -- {id: llm.embeddings.azure_openai.basic.nonstream.works, module: llm, tier: P0, subject_endpoint: embeddings, route: azure_openai, capability: basic, streaming: nonstream, assertions: [works], source: "llms/azure/azure.py", rationale: "Azure embeddings via translation"} -- {id: llm.embeddings.bedrock.basic.nonstream.works, module: llm, tier: P1, subject_endpoint: embeddings, route: bedrock_converse, capability: basic, streaming: nonstream, assertions: [works], source: "llms/bedrock/embed/embedding.py", rationale: "Bedrock Titan embeddings"} -- {id: llm.embeddings.vertex.basic.nonstream.works, module: llm, tier: P1, subject_endpoint: embeddings, route: vertex, capability: basic, streaming: nonstream, assertions: [works], source: "vertex_embeddings/embedding_handler.py", rationale: "Vertex embeddings"} -- {id: llm.embeddings.cohere.basic.nonstream.works, module: llm, tier: P1, subject_endpoint: embeddings, route: cohere, capability: basic, streaming: nonstream, assertions: [works], source: "llms/cohere/embed/handler.py", rationale: "Cohere embeddings"} -- {id: llm.embeddings.anthropic.basic.nonstream.works, module: llm, tier: P1, subject_endpoint: embeddings, route: anthropic, capability: basic, streaming: nonstream, assertions: [works], source: "llms/anthropic/chat/handler.py", rationale: "Anthropic vector API (verify support)"} -- {id: llm.batches.openai.create.nonstream.works, module: llm, tier: P0, subject_endpoint: batches, route: openai, capability: basic, streaming: nonstream, assertions: [works], source: "test_batches_e2e.py", rationale: "Core batch create"} -- {id: llm.batches.openai.retrieve.nonstream.works, module: llm, tier: P0, subject_endpoint: batches, route: openai, capability: basic, streaming: nonstream, assertions: [works], source: "test_batches_e2e.py", rationale: "Batch retrieve, id round-trip + status"} -- {id: llm.batches.openai.cancel.nonstream.works, module: llm, tier: P0, subject_endpoint: batches, route: openai, capability: basic, streaming: nonstream, assertions: [works], source: "test_batches_e2e.py", rationale: "Batch cancel"} -- {id: llm.batches.openai.list.nonstream.works, module: llm, tier: P0, subject_endpoint: batches, route: openai, capability: basic, streaming: nonstream, assertions: [works], source: "test_batches_e2e.py", rationale: "Batch list envelope"} -- {id: llm.batches.openai.file_lifecycle.nonstream.works, module: llm, tier: P0, subject_endpoint: batches, route: openai, capability: basic, streaming: nonstream, assertions: [works], source: "test_batches_e2e.py", rationale: "File upload/retrieve/delete for batch flow"} -- {id: llm.batches.openai_encoded.basic.nonstream.works, module: llm, tier: P0, subject_endpoint: batches, route: openai, capability: basic, streaming: nonstream, assertions: [works], source: "batches/capabilities.py", rationale: "Encoded scenario lifecycle"} -- {id: llm.batches.openai_unified.basic.nonstream.works, module: llm, tier: P0, subject_endpoint: batches, route: openai, capability: basic, streaming: nonstream, assertions: [works], source: "batches/capabilities.py", rationale: "Unified/managed-id scenario"} -- {id: llm.batches.openai_model_param.basic.nonstream.works, module: llm, tier: P0, subject_endpoint: batches, route: openai, capability: basic, streaming: nonstream, assertions: [works], source: "batches/capabilities.py", rationale: "Model-param scenario"} -- {id: llm.batches.openai_provider_fallback.basic.nonstream.works, module: llm, tier: P0, subject_endpoint: batches, route: openai, capability: basic, streaming: nonstream, assertions: [works], source: "batches/capabilities.py", rationale: "Provider-fallback raw-id scenario"} -- {id: llm.batches.azure_openai.basic.nonstream.works, module: llm, tier: P0, subject_endpoint: batches, route: azure_openai, capability: basic, streaming: nonstream, assertions: [works], source: "batches/capabilities.py:98", rationale: "Azure batches all scenarios"} -- {id: llm.batches.vertex.basic.nonstream.works, module: llm, tier: P0, subject_endpoint: batches, route: vertex, capability: basic, streaming: nonstream, assertions: [works], source: "batches/capabilities.py:98", rationale: "Vertex batches"} -- {id: llm.batches.bedrock.basic.nonstream.works, module: llm, tier: P0, subject_endpoint: batches, route: bedrock_converse, capability: basic, streaming: nonstream, assertions: [works], source: "batches/capabilities.py:98", rationale: "Bedrock batches (encoded/unified only)"} -- {id: llm.batches.openai.key_model_access_denied.nonstream.works, module: llm, tier: P0, subject_endpoint: batches, route: openai, capability: basic, streaming: nonstream, assertions: [works], source: "test_batches_e2e.py", rationale: "Key model restriction 403 on upload/create"} -- {id: llm.files.openai.upload.nonstream.works, module: llm, tier: P0, subject_endpoint: files, route: openai, capability: basic, streaming: nonstream, assertions: [works], source: "openai_files_endpoints/files_endpoints.py:46", rationale: "File upload returns OpenAIFileObject"} -- {id: llm.files.openai.retrieve.nonstream.works, module: llm, tier: P0, subject_endpoint: files, route: openai, capability: basic, streaming: nonstream, assertions: [works], source: "files_endpoints.py", rationale: "File retrieve by id"} -- {id: llm.files.openai.delete.nonstream.works, module: llm, tier: P0, subject_endpoint: files, route: openai, capability: basic, streaming: nonstream, assertions: [works], source: "files_endpoints.py", rationale: "File delete returns deleted=true"} -- {id: llm.files.openai.list.nonstream.works, module: llm, tier: P0, subject_endpoint: files, route: openai, capability: basic, streaming: nonstream, assertions: [works], source: "files_endpoints.py", rationale: "File list paginated"} -- {id: llm.files.azure_openai.upload.nonstream.works, module: llm, tier: P0, subject_endpoint: files, route: azure_openai, capability: basic, streaming: nonstream, assertions: [works], source: "batches/capabilities.py:45", rationale: "Azure file upload managed backend"} -- {id: llm.files.vertex.upload.nonstream.works, module: llm, tier: P0, subject_endpoint: files, route: vertex, capability: basic, streaming: nonstream, assertions: [works], source: "batches/capabilities.py:52", rationale: "Vertex file upload to GCS"} -- {id: llm.files.bedrock.upload.nonstream.works, module: llm, tier: P0, subject_endpoint: files, route: bedrock_converse, capability: basic, streaming: nonstream, assertions: [works], source: "batches/capabilities.py:59", rationale: "Bedrock file upload to S3"} -- {id: llm.rerank.cohere.basic.nonstream.works, module: llm, tier: P1, subject_endpoint: rerank, route: cohere, capability: basic, streaming: nonstream, assertions: [works], source: "test_rerank_e2e.py:29", rationale: "Cohere rerank, top_n + relevance_score"} -- {id: llm.rerank.bedrock.basic.nonstream.works, module: llm, tier: P1, subject_endpoint: rerank, route: bedrock_converse, capability: basic, streaming: nonstream, assertions: [works], source: "llms/bedrock/rerank/handler.py", rationale: "Bedrock rerank"} -- {id: llm.rerank.together_ai.basic.nonstream.works, module: llm, tier: P1, subject_endpoint: rerank, route: together_ai, capability: basic, streaming: nonstream, assertions: [works], source: "llms/together_ai/rerank/handler.py", rationale: "Together rerank"} -- {id: llm.images_generations.openai.basic.nonstream.works, module: llm, tier: P1, subject_endpoint: images_generations, route: openai, capability: basic, streaming: nonstream, assertions: [works], source: "test_image_generation_e2e.py:22", rationale: "OpenAI image gen, b64/url"} -- {id: llm.images_generations.azure_openai.basic.nonstream.works, module: llm, tier: P1, subject_endpoint: images_generations, route: azure_openai, capability: basic, streaming: nonstream, assertions: [works], source: "llms/azure/azure.py", rationale: "Azure DALL-E"} -- {id: llm.images_generations.vertex.basic.nonstream.works, module: llm, tier: P1, subject_endpoint: images_generations, route: vertex, capability: basic, streaming: nonstream, assertions: [works], source: "vertex_ai/image_generation/image_generation_handler.py", rationale: "Vertex Imagen"} -- {id: llm.images_generations.bedrock.basic.nonstream.works, module: llm, tier: P1, subject_endpoint: images_generations, route: bedrock_converse, capability: basic, streaming: nonstream, assertions: [works], source: "bedrock/image_generation/image_handler.py", rationale: "Bedrock Titan Image"} -- {id: llm.images_generations.black_forest_labs.basic.nonstream.works, module: llm, tier: P1, subject_endpoint: images_generations, route: openai, capability: basic, streaming: nonstream, assertions: [works], source: "black_forest_labs/image_generation/handler.py", rationale: "BFL Flux via OpenAI-compat"} -- {id: llm.audio_speech.openai.basic.nonstream.works, module: llm, tier: P1, subject_endpoint: audio_speech, route: openai, capability: basic, streaming: nonstream, assertions: [works], source: "test_audio_speech_e2e.py:22", rationale: "OpenAI TTS binary audio"} -- {id: llm.audio_speech.openai.basic.stream.works, module: llm, tier: P1, subject_endpoint: audio_speech, route: openai, capability: basic, streaming: stream, assertions: [works], source: "proxy_server.py:9043", rationale: "TTS streaming chunk generator"} -- {id: llm.audio_speech.azure_openai.basic.nonstream.works, module: llm, tier: P1, subject_endpoint: audio_speech, route: azure_openai, capability: basic, streaming: nonstream, assertions: [works], source: "llms/azure/azure.py", rationale: "Azure TTS"} -- {id: llm.audio_speech.vertex.basic.nonstream.works, module: llm, tier: P1, subject_endpoint: audio_speech, route: vertex, capability: basic, streaming: nonstream, assertions: [works], source: "vertex_ai/text_to_speech/text_to_speech_handler.py", rationale: "Vertex TTS"} -- {id: llm.audio_transcriptions.openai.basic.nonstream.works, module: llm, tier: P1, subject_endpoint: audio_transcriptions, route: openai, capability: basic, streaming: nonstream, assertions: [works], source: "openai/transcriptions/handler.py", rationale: "OpenAI Whisper"} -- {id: llm.audio_transcriptions.azure_openai.basic.nonstream.works, module: llm, tier: P1, subject_endpoint: audio_transcriptions, route: azure_openai, capability: basic, streaming: nonstream, assertions: [works], source: "azure/audio_transcriptions.py", rationale: "Azure STT"} -- {id: llm.audio_transcriptions.soniox.basic.nonstream.works, module: llm, tier: P2, subject_endpoint: audio_transcriptions, route: openai, capability: basic, streaming: nonstream, assertions: [works], source: "soniox/audio_transcription/handler.py", rationale: "Soniox via OpenAI-compat (smoke)"} -- {id: llm.audio_transcriptions.nvidia_riva.basic.nonstream.works, module: llm, tier: P2, subject_endpoint: audio_transcriptions, route: openai, capability: basic, streaming: nonstream, assertions: [works], source: "nvidia_riva/audio_transcription/handler.py", rationale: "NVIDIA Riva (smoke)"} -- {id: llm.moderations.openai.basic.nonstream.works, module: llm, tier: P1, subject_endpoint: moderations, route: openai, capability: basic, streaming: nonstream, assertions: [works], source: "proxy_server.py", rationale: "OpenAI moderations (only provider)"} diff --git a/tests/e2e/coverage_registry/logging.yaml b/tests/e2e/coverage_registry/logging.yaml deleted file mode 100644 index 65ab8f0096f..00000000000 --- a/tests/e2e/coverage_registry/logging.yaml +++ /dev/null @@ -1,25 +0,0 @@ -# Logging integration delivery (behavior features). Grounded in litellm/integrations/. -- {id: logging.langfuse.success.logs_spend, module: logging, tier: P0, event: success, assertions: [logs_spend], exercised_on: [chat_completions, messages, embeddings], source: "integrations/langfuse/langfuse.py", rationale: "Primary tracing backend; cost accuracy"} -- {id: logging.langfuse.failure.logs_spend, module: logging, tier: P0, event: failure, assertions: [logs_spend], exercised_on: [chat_completions, messages], source: "integrations/langfuse/langfuse.py", rationale: "Failure path must still track spend"} -- {id: logging.langfuse.stream.logs_spend, module: logging, tier: P0, event: stream, assertions: [logs_spend], exercised_on: [chat_completions, messages], source: "integrations/langfuse/langfuse.py", rationale: "Streaming token counts aggregate"} -- {id: logging.s3.success.writes_object, module: logging, tier: P0, event: success, assertions: [writes_object], exercised_on: [chat_completions, messages, embeddings], source: "integrations/s3_v2.py", rationale: "Primary audit trail; batch flush no-drop"} -- {id: logging.s3.failure.writes_object, module: logging, tier: P0, event: failure, assertions: [writes_object], exercised_on: [chat_completions, messages], source: "integrations/s3_v2.py", rationale: "Failed calls persisted for compliance"} -- {id: logging.gcs_bucket.success.writes_object, module: logging, tier: P0, event: success, assertions: [writes_object], exercised_on: [chat_completions, messages, embeddings], source: "integrations/gcs_bucket/gcs_bucket.py", rationale: "GCS parallel to S3"} -- {id: logging.datadog.success.exports_metric, module: logging, tier: P0, event: success, assertions: [exports_metric], exercised_on: [chat_completions, messages, embeddings], source: "integrations/datadog/datadog.py", rationale: "Powers dashboards/alerts; cardinality regressions common"} -- {id: logging.datadog.failure.exports_metric, module: logging, tier: P0, event: failure, assertions: [exports_metric], exercised_on: [chat_completions], source: "integrations/datadog/datadog.py", rationale: "Failure metrics for alerting/SLO"} -- {id: logging.prometheus.success.exports_metric, module: logging, tier: P0, event: success, assertions: [exports_metric], exercised_on: [chat_completions, messages, embeddings], source: "integrations/prometheus.py", rationale: "Standard OSS metrics; per-key cardinality (existing e2e)"} -- {id: logging.otel.success.exports_metric, module: logging, tier: P0, event: success, assertions: [exports_metric], exercised_on: [chat_completions, messages, embeddings], source: "integrations/otel/logger.py", rationale: "OTEL spans on every call path"} -- {id: logging.otel.failure.exports_metric, module: logging, tier: P0, event: failure, assertions: [exports_metric], exercised_on: [chat_completions, messages], source: "integrations/otel/logger.py", rationale: "Error spans for observability continuity"} -- {id: logging.braintrust.success.logs_spend, module: logging, tier: P1, event: success, assertions: [logs_spend], exercised_on: [chat_completions, messages], source: "integrations/braintrust_logging.py", rationale: "Evals platform spend"} -- {id: logging.langsmith.success.logs_spend, module: logging, tier: P1, event: success, assertions: [logs_spend], exercised_on: [chat_completions, messages], source: "integrations/langsmith.py", rationale: "LangChain ecosystem"} -- {id: logging.arize.success.logs_spend, module: logging, tier: P1, event: success, assertions: [logs_spend], exercised_on: [chat_completions, embeddings], source: "integrations/arize/arize.py", rationale: "ML-ops observability"} -- {id: logging.mlflow.success.logs_spend, module: logging, tier: P1, event: success, assertions: [logs_spend], exercised_on: [chat_completions], source: "integrations/mlflow.py", rationale: "Experiment tracking cost/run"} -- {id: logging.opik.success.logs_spend, module: logging, tier: P1, event: success, assertions: [logs_spend], exercised_on: [chat_completions], source: "integrations/opik/opik.py", rationale: "Eval platform spend/case"} -- {id: logging.openmeter.success.exports_metric, module: logging, tier: P1, event: success, assertions: [exports_metric], exercised_on: [chat_completions, messages, embeddings], source: "integrations/openmeter.py", rationale: "Usage metering for billing"} -- {id: logging.literal_ai.success.logs_spend, module: logging, tier: P1, event: success, assertions: [logs_spend], exercised_on: [chat_completions, messages], source: "integrations/literal_ai.py", rationale: "Tracing platform spend"} -- {id: logging.posthog.success.exports_metric, module: logging, tier: P1, event: success, assertions: [exports_metric], exercised_on: [chat_completions, messages], source: "integrations/posthog.py", rationale: "Product analytics batching"} -- {id: logging.azure_storage.success.writes_object, module: logging, tier: P1, event: success, assertions: [writes_object], exercised_on: [chat_completions, messages], source: "integrations/azure_storage/azure_storage.py", rationale: "Azure blob for enterprise"} -- {id: logging.cloudzero.success.logs_spend, module: logging, tier: P1, event: success, assertions: [logs_spend], exercised_on: [chat_completions], source: "integrations/cloudzero/cloudzero.py", rationale: "Cost ops correlation"} -- {id: logging.focus.success.writes_object, module: logging, tier: P1, event: success, assertions: [writes_object], exercised_on: [chat_completions, messages], source: "integrations/focus/focus_logger.py", rationale: "Cost mgmt multi-destination export"} -- {id: logging.niche_integrations.success.logs_spend, module: logging, tier: P2, event: success, assertions: [logs_spend], exercised_on: [chat_completions], source: grammar, rationale: "SMOKE cohort: athina/galileo/deepeval/langtrace/weave/lunary/humanloop/traceloop/helicone/argilla/newrelic/sqs/supabase/dynamodb/agentops/lago/etc"} -- {id: logging.niche_integrations.failure.logs_spend, module: logging, tier: P2, event: failure, assertions: [logs_spend], exercised_on: [chat_completions], source: grammar, rationale: "SMOKE niche failure path"} diff --git a/tests/e2e/coverage_registry/mcp.yaml b/tests/e2e/coverage_registry/mcp.yaml deleted file mode 100644 index d477b257cb0..00000000000 --- a/tests/e2e/coverage_registry/mcp.yaml +++ /dev/null @@ -1,113 +0,0 @@ -# MCP module. Grounded in litellm/proxy/_experimental/mcp_server/. See tests/e2e/CLAUDE.md for the grammar. -- id: mcp.list_tools.api_key.succeeds - module: mcp - tier: P0 - operation: list_tools - auth_family: api_key - assertions: [succeeds] - source: "server.py:637" - rationale: Core operation; most common auth path; high usage -- id: mcp.list_tools.api_key.denied_without_permission - module: mcp - tier: P0 - operation: list_tools - auth_family: api_key - assertions: [denied_without_permission] - source: "mcp_server_manager.py:1409" - rationale: Permission guard is high blast-radius; multi-tenant safety -- id: mcp.call_tool.api_key.succeeds - module: mcp - tier: P0 - operation: call_tool - auth_family: api_key - assertions: [succeeds] - source: "server.py:849" - rationale: Primary operation; customer-critical; high usage -- id: mcp.call_tool.api_key.denied_without_permission - module: mcp - tier: P0 - operation: call_tool - auth_family: api_key - assertions: [denied_without_permission] - source: "rest_endpoints.py:305-386" - rationale: Tool-level permission guard; multi-tenant safety -- id: mcp.list_tools.bearer.succeeds - module: mcp - tier: P1 - operation: list_tools - auth_family: bearer - assertions: [succeeds] - source: "server.py:662" - rationale: OAuth/bearer token flow; upstream delegation -- id: mcp.call_tool.bearer.succeeds - module: mcp - tier: P1 - operation: call_tool - auth_family: bearer - assertions: [succeeds] - source: "server.py:886" - rationale: Bearer token forwarding for tool invocation -- id: mcp.list_tools.oauth.succeeds - module: mcp - tier: P1 - operation: list_tools - auth_family: oauth - assertions: [succeeds] - source: "rest_endpoints.py:138-188" - rationale: Interactive OAuth2 flow; live token management -- id: mcp.call_tool.oauth.succeeds - module: mcp - tier: P1 - operation: call_tool - auth_family: oauth - assertions: [succeeds] - source: "db.py user_oauth_credential lookup" - rationale: OAuth2 token passthrough; per-user credential storage -- id: mcp.list_tools.none.succeeds - module: mcp - tier: P1 - operation: list_tools - auth_family: none - assertions: [succeeds] - source: "mcp_server_manager.py:1485-1492" - rationale: Public/anonymous servers; delegate_auth_to_upstream -- id: mcp.call_tool.none.succeeds - module: mcp - tier: P1 - operation: call_tool - auth_family: none - assertions: [succeeds] - source: "rest_endpoints.py:305-334" - rationale: No upstream auth required; demo servers -- id: mcp.get_prompt.api_key.succeeds - module: mcp - tier: P1 - operation: get_prompt - auth_family: api_key - assertions: [succeeds] - source: "server.py:1042" - rationale: Prompt op; same auth stack as tools -- id: mcp.read_resource.api_key.succeeds - module: mcp - tier: P1 - operation: read_resource - auth_family: api_key - assertions: [succeeds] - source: "server.py:1177" - rationale: Resource op; same permission model as tools -- id: mcp.list_prompts.api_key.succeeds - module: mcp - tier: P2 - operation: list_prompts - auth_family: api_key - assertions: [succeeds] - source: "server.py:993" - rationale: Smoke-level; same auth stack as list_tools -- id: mcp.list_resources.api_key.succeeds - module: mcp - tier: P2 - operation: list_resources - auth_family: api_key - assertions: [succeeds] - source: "server.py:1089" - rationale: Smoke; rarely used; same auth model as tools diff --git a/tests/e2e/coverage_registry/mgmt.yaml b/tests/e2e/coverage_registry/mgmt.yaml deleted file mode 100644 index 4fbaac0a205..00000000000 --- a/tests/e2e/coverage_registry/mgmt.yaml +++ /dev/null @@ -1,68 +0,0 @@ -# Management/UI endpoint features. Grounded in litellm/proxy/management_endpoints/. -- {id: mgmt.key.generate.persists, module: mgmt, tier: P0, surface: api, assertions: [persists], source: "key_management_endpoints.py:1444", rationale: "API key survives DB roundtrip"} -- {id: mgmt.key.generate.admin_only, module: mgmt, tier: P0, surface: api, assertions: [admin_only], source: "key_management_endpoints.py:1444", rationale: "Only master/team-admin creates keys"} -- {id: mgmt.key.generate.happy_path, module: mgmt, tier: P0, surface: ui, assertions: [happy_path], source: "ui_sso.py:420", rationale: "SSO-driven key gen (UI path)"} -- {id: mgmt.key.update.persists, module: mgmt, tier: P0, surface: api, assertions: [persists], source: "key_management_endpoints.py:2462", rationale: "Budget/model changes persist"} -- {id: mgmt.key.update.admin_only, module: mgmt, tier: P0, surface: api, assertions: [admin_only], source: "key_management_endpoints.py:2462", rationale: "Non-admin cannot escalate perms"} -- {id: mgmt.key.update.happy_path, module: mgmt, tier: P1, surface: ui, assertions: [happy_path], source: "key_management_endpoints.py:2462", rationale: "Key edit through the dashboard"} -- {id: mgmt.key.delete.persists, module: mgmt, tier: P0, surface: api, assertions: [persists], source: "key_management_endpoints.py:3122", rationale: "Deletion revokes future calls"} -- {id: mgmt.key.delete.admin_only, module: mgmt, tier: P0, surface: api, assertions: [admin_only], source: "key_management_endpoints.py:3122", rationale: "Non-owner cannot delete"} -- {id: mgmt.key.info.persists, module: mgmt, tier: P0, surface: api, assertions: [persists], source: "key_management_endpoints.py:3380", rationale: "Info reflects all writes"} -- {id: mgmt.team.new.persists, module: mgmt, tier: P0, surface: api, assertions: [persists], source: "team_endpoints.py:897", rationale: "team_id/alias/budgets stored"} -- {id: mgmt.team.new.admin_only, module: mgmt, tier: P0, surface: api, assertions: [admin_only], source: "team_endpoints.py:897", rationale: "Only org-admin/master creates teams"} -- {id: mgmt.team.member_add.persists, module: mgmt, tier: P0, surface: api, assertions: [persists], source: "team_endpoints.py:2424", rationale: "Membership + per-member budget persist"} -- {id: mgmt.team.member_add.member_forbidden, module: mgmt, tier: P0, surface: api, assertions: [member_forbidden], source: "team_endpoints.py:2424", rationale: "Non-admin forbidden to add"} -- {id: mgmt.team.member_delete.persists, module: mgmt, tier: P0, surface: api, assertions: [persists], source: "team_endpoints.py:2800", rationale: "Removal revokes team key access"} -- {id: mgmt.team.member_delete.member_forbidden, module: mgmt, tier: P0, surface: api, assertions: [member_forbidden], source: "team_endpoints.py:2800", rationale: "Non-admin forbidden to remove"} -- {id: mgmt.budget.new.persists, module: mgmt, tier: P0, surface: api, assertions: [persists], source: "budget_management_endpoints.py:40", rationale: "max/soft/reset windows persist"} -- {id: mgmt.budget.new.admin_only, module: mgmt, tier: P0, surface: api, assertions: [admin_only], source: "budget_management_endpoints.py:40", rationale: "Requires master/admin"} -- {id: mgmt.model.add.persists, module: mgmt, tier: P0, surface: api, assertions: [persists], source: "model_management_endpoints.py:1201", rationale: "Registration persists for routing"} -- {id: mgmt.model.add.admin_only, module: mgmt, tier: P0, surface: api, assertions: [admin_only], source: "model_management_endpoints.py:1201", rationale: "Non-admin cannot inject model config"} -- {id: mgmt.user.new.happy_path, module: mgmt, tier: P0, surface: api, assertions: [happy_path], source: "internal_user_endpoints.py:360", rationale: "User creation full cycle"} -- {id: mgmt.key.list.happy_path, module: mgmt, tier: P1, surface: api, assertions: [happy_path], source: "key_management_endpoints.py:5119", rationale: "Key inventory pagination"} -- {id: mgmt.key.block.persists, module: mgmt, tier: P1, surface: api, assertions: [persists], source: "key_management_endpoints.py:5849", rationale: "Blocked stays blocked on restart"} -- {id: mgmt.key.unblock.persists, module: mgmt, tier: P1, surface: api, assertions: [persists], source: "key_management_endpoints.py:5960", rationale: "Unblock restores access"} -- {id: mgmt.key.regenerate.happy_path, module: mgmt, tier: P1, surface: api, assertions: [happy_path], source: "key_management_endpoints.py:6071", rationale: "Rotation: new works, old invalid"} -- {id: mgmt.key.health.happy_path, module: mgmt, tier: P1, surface: api, assertions: [happy_path], source: "key_management_endpoints.py:4292", rationale: "Key health endpoint"} -- {id: mgmt.key.bulk_update.happy_path, module: mgmt, tier: P1, surface: api, assertions: [happy_path], source: "key_management_endpoints.py:2677", rationale: "Batch key updates"} -- {id: mgmt.team.update.persists, module: mgmt, tier: P1, surface: api, assertions: [persists], source: "team_endpoints.py:1582", rationale: "Metadata/budget updates persist"} -- {id: mgmt.team.delete.persists, module: mgmt, tier: P1, surface: api, assertions: [persists], source: "team_endpoints.py:1750", rationale: "Deletion prevents key access"} -- {id: mgmt.team.block.persists, module: mgmt, tier: P1, surface: api, assertions: [persists], source: "team_endpoints.py", rationale: "Block suspends all members"} -- {id: mgmt.team.info.happy_path, module: mgmt, tier: P1, surface: api, assertions: [happy_path], source: "team_endpoints.py:2244", rationale: "Metadata+members+budgets"} -- {id: mgmt.team.list.happy_path, module: mgmt, tier: P1, surface: api, assertions: [happy_path], source: "team_endpoints.py:3645", rationale: "Pagination/filtering"} -- {id: mgmt.team.member_update.persists, module: mgmt, tier: P1, surface: api, assertions: [persists], source: "team_endpoints.py:2768", rationale: "Member budget/role updates persist"} -- {id: mgmt.user.update.persists, module: mgmt, tier: P1, surface: api, assertions: [persists], source: "internal_user_endpoints.py:555", rationale: "Metadata/perm updates persist"} -- {id: mgmt.user.delete.persists, module: mgmt, tier: P1, surface: api, assertions: [persists], source: "internal_user_endpoints.py:640", rationale: "Deletion revokes keys+teams"} -- {id: mgmt.user.list.happy_path, module: mgmt, tier: P1, surface: api, assertions: [happy_path], source: "internal_user_endpoints.py:475", rationale: "Admin view all users"} -- {id: mgmt.user.info.happy_path, module: mgmt, tier: P1, surface: api, assertions: [happy_path], source: "internal_user_endpoints.py:440", rationale: "Roles/perms/team membership"} -- {id: mgmt.organization.new.happy_path, module: mgmt, tier: P1, surface: api, assertions: [happy_path], source: "organization_endpoints.py:403", rationale: "Org for multi-tenant isolation"} -- {id: mgmt.organization.update.persists, module: mgmt, tier: P1, surface: api, assertions: [persists], source: "organization_endpoints.py:545", rationale: "Org metadata updates persist"} -- {id: mgmt.organization.delete.persists, module: mgmt, tier: P1, surface: api, assertions: [persists], source: "organization_endpoints.py:710", rationale: "Cascades to teams/keys"} -- {id: mgmt.organization.member_add.happy_path, module: mgmt, tier: P1, surface: api, assertions: [happy_path], source: "organization_endpoints.py:835", rationale: "Org member onboarding"} -- {id: mgmt.customer.new.happy_path, module: mgmt, tier: P1, surface: api, assertions: [happy_path], source: "customer_endpoints.py:372", rationale: "End-user for spend tracking"} -- {id: mgmt.customer.delete.persists, module: mgmt, tier: P1, surface: api, assertions: [persists], source: "customer_endpoints.py:480", rationale: "Removes from spend tracking"} -- {id: mgmt.end_user.new.happy_path, module: mgmt, tier: P1, surface: api, assertions: [happy_path], source: "customer_endpoints.py:730", rationale: "End-user create (synonym)"} -- {id: mgmt.tag.new.happy_path, module: mgmt, tier: P1, surface: api, assertions: [happy_path], source: "tag_management_endpoints.py:160", rationale: "Tag for spend categorization"} -- {id: mgmt.tag.list.happy_path, module: mgmt, tier: P1, surface: api, assertions: [happy_path], source: "tag_management_endpoints.py:315", rationale: "Tag enumeration"} -- {id: mgmt.tag.delete.persists, module: mgmt, tier: P1, surface: api, assertions: [persists], source: "tag_management_endpoints.py:390", rationale: "Stops future tagging"} -- {id: mgmt.model.update.persists, module: mgmt, tier: P1, surface: api, assertions: [persists], source: "model_management_endpoints.py:1358", rationale: "Pricing/concurrency persist"} -- {id: mgmt.model.delete.persists, module: mgmt, tier: P1, surface: api, assertions: [persists], source: "model_management_endpoints.py:1045", rationale: "Removes from registry"} -- {id: mgmt.model.block.persists, module: mgmt, tier: P1, surface: api, assertions: [persists], source: "model_management_endpoints.py", rationale: "Blocked model stays blocked"} -- {id: mgmt.access_group.new.happy_path, module: mgmt, tier: P1, surface: api, assertions: [happy_path], source: "model_access_group_management_endpoints.py:450", rationale: "Model permissioning group"} -- {id: mgmt.access_group.info.happy_path, module: mgmt, tier: P1, surface: api, assertions: [happy_path], source: "model_access_group_management_endpoints.py:600", rationale: "Access group membership query"} -- {id: mgmt.mcp_server.register.happy_path, module: mgmt, tier: P1, surface: api, assertions: [happy_path], source: "mcp_management_endpoints.py:880", rationale: "MCP server registration"} -- {id: mgmt.mcp_server.approve.persists, module: mgmt, tier: P1, surface: api, assertions: [persists], source: "mcp_management_endpoints.py:1200", rationale: "Admin approval persists"} -- {id: mgmt.budget.update.persists, module: mgmt, tier: P1, surface: api, assertions: [persists], source: "budget_management_endpoints.py:155", rationale: "Limit changes apply"} -- {id: mgmt.budget.delete.persists, module: mgmt, tier: P1, surface: api, assertions: [persists], source: "budget_management_endpoints.py:280", rationale: "Clears limits"} -- {id: mgmt.budget.list.happy_path, module: mgmt, tier: P1, surface: api, assertions: [happy_path], source: "budget_management_endpoints.py:215", rationale: "Budget enumeration"} -- {id: mgmt.callback.list.happy_path, module: mgmt, tier: P2, surface: api, assertions: [happy_path], source: "callback_management_endpoints.py", rationale: "Callback config (smoke)"} -- {id: mgmt.cache_settings.update.happy_path, module: mgmt, tier: P2, surface: api, assertions: [happy_path], source: "cache_settings_endpoints.py", rationale: "Cache config (smoke)"} -- {id: mgmt.cost_tracking.estimate.happy_path, module: mgmt, tier: P2, surface: api, assertions: [happy_path], source: "cost_tracking_settings.py", rationale: "Cost estimate (smoke)"} -- {id: mgmt.router_settings.update.happy_path, module: mgmt, tier: P2, surface: api, assertions: [happy_path], source: "router_settings_endpoints.py", rationale: "Router config (smoke)"} -- {id: mgmt.jwt_key_mapping.new.happy_path, module: mgmt, tier: P2, surface: api, assertions: [happy_path], source: "jwt_key_mapping_endpoints.py", rationale: "JWT->key mapping (smoke)"} -- {id: mgmt.compliance.gdpr.happy_path, module: mgmt, tier: P2, surface: api, assertions: [happy_path], source: "compliance_endpoints.py", rationale: "GDPR ops (smoke)"} -- {id: mgmt.tool_management.list.happy_path, module: mgmt, tier: P2, surface: api, assertions: [happy_path], source: "tool_management_endpoints.py", rationale: "Tool inventory (smoke)"} -- {id: mgmt.fallback_management.update.happy_path, module: mgmt, tier: P2, surface: api, assertions: [happy_path], source: "fallback_management_endpoints.py", rationale: "Fallback config (smoke)"} -- {id: mgmt.config_override.hashicorp_vault.happy_path, module: mgmt, tier: P2, surface: api, assertions: [happy_path], source: "config_override_endpoints.py", rationale: "Vault integration (smoke)"} -- {id: mgmt.workflow.list.happy_path, module: mgmt, tier: P2, surface: api, assertions: [happy_path], source: "workflow_management_endpoints.py", rationale: "Workflow tracking (smoke)"} -- {id: mgmt.credential_migration.check.happy_path, module: mgmt, tier: P2, surface: api, assertions: [happy_path], source: "key_management_endpoints.py:4252", rationale: "Encryption migration (smoke)"} diff --git a/tests/e2e/coverage_registry/other.yaml b/tests/e2e/coverage_registry/other.yaml deleted file mode 100644 index c2efecec677..00000000000 --- a/tests/e2e/coverage_registry/other.yaml +++ /dev/null @@ -1,28 +0,0 @@ -# Other (holding pen). Grounded in litellm/proxy/auth/ + health_endpoints/ + proxy_server.py. -# PROMOTION NOTE: the auth cluster (~14 cells) is a candidate to promote to its own module once stable. -- {id: other.auth.master_key.valid_allows, module: other, tier: P0, area: auth, assertions: [valid_allows], source: "user_api_key_auth.py:1569-1588", rationale: "Master key authenticates; timing-safe compare"} -- {id: other.auth.master_key.invalid_denied, module: other, tier: P0, area: auth, assertions: [invalid_denied], source: "user_api_key_auth.py:1580", rationale: "Invalid master key rejected"} -- {id: other.auth.jwt.valid_token_allows, module: other, tier: P0, area: auth, assertions: [valid_token_allows], source: "handle_jwt.py:77-150", rationale: "Valid JWT with correct issuer + claims grants access"} -- {id: other.auth.jwt.expired_denied, module: other, tier: P0, area: auth, assertions: [expired_denied], source: "handle_jwt.py:125-135", rationale: "Expired JWT rejected even with valid signature"} -- {id: other.auth.jwt.invalid_signature_denied, module: other, tier: P0, area: auth, assertions: [invalid_signature_denied], source: "handle_jwt.py:145-150", rationale: "Bad/missing signature fails verification"} -- {id: other.auth.virtual_key.route_permission_enforced, module: other, tier: P0, area: auth, assertions: [route_permission_enforced], source: "route_checks.py:89-151", rationale: "allowed_routes whitelist denies disallowed routes"} -- {id: other.auth.virtual_key.route_group_allowed, module: other, tier: P1, area: auth, assertions: [route_group_allowed], source: "route_checks.py:106-128", rationale: "allowed_routes=[llm_api_routes] grants all LLM endpoints"} -- {id: other.auth.passthrough.model_allowlist_enforced, module: other, tier: P1, area: auth, assertions: [model_allowlist_enforced], source: "route_checks.py:135-151", rationale: "Passthrough enforces per-key model allow-lists"} -- {id: other.auth.oauth2.token_valid_allows, module: other, tier: P1, area: auth, assertions: [token_valid_allows], source: "oauth2_check.py:15-73", rationale: "OAuth2 introspection grants active token"} -- {id: other.auth.oauth2.token_invalid_denied, module: other, tier: P1, area: auth, assertions: [token_invalid_denied], source: "oauth2_check.py:37-73", rationale: "Expired/inactive OAuth2 token denied"} -- {id: other.auth.ip_allowlist.internal_ip_allows, module: other, tier: P1, area: auth, assertions: [internal_ip_allows], source: "ip_address_utils.py:54-76", rationale: "Internal CIDR bypasses public-API restriction"} -- {id: other.auth.ip_allowlist.external_ip_denied_to_private, module: other, tier: P1, area: auth, assertions: [external_ip_denied_to_private], source: "ip_address_utils.py:54-76", rationale: "External IP cannot reach internal-only resources"} -- {id: other.lifecycle.readiness.public_probe, module: other, tier: P0, area: lifecycle, assertions: [public_probe], source: "_health_endpoints.py:1551-1570", rationale: "Unauthenticated /health/readiness safe for LBs"} -- {id: other.lifecycle.readiness.reports_db_status, module: other, tier: P0, area: lifecycle, assertions: [reports_db_status], source: "_health_endpoints.py:1551-1570", rationale: "readiness distinguishes healthy vs DB-unreachable"} -- {id: other.lifecycle.readiness.shutting_down_returns_503, module: other, tier: P0, area: lifecycle, assertions: [shutting_down_returns_503], source: "_health_endpoints.py:1554-1556", rationale: "Graceful shutdown drains LB via 503"} -- {id: other.lifecycle.readiness_details.authenticated_diagnostics, module: other, tier: P1, area: lifecycle, assertions: [authenticated_diagnostics], source: "_health_endpoints.py:1574-1584", rationale: "Auth'd details expose cache/callback status"} -- {id: other.lifecycle.liveness.ping, module: other, tier: P1, area: lifecycle, assertions: [ping], source: "_health_endpoints.py:134-155", rationale: "Liveness confirms server responding"} -- {id: other.lifecycle.startup.config_loads, module: other, tier: P0, area: lifecycle, assertions: [config_loads], source: "proxy_server.py:4020-4100", rationale: "Startup loads YAML, resolves env, persists to DB"} -- {id: other.lifecycle.startup.env_vars_resolved, module: other, tier: P1, area: lifecycle, assertions: [env_vars_resolved], source: "proxy_server.py:3984-4010", rationale: "os.environ/ refs resolved at startup"} -- {id: other.lifecycle.background_health_check.interval_configurable, module: other, tier: P1, area: lifecycle, assertions: [interval_configurable], source: "proxy_server.py:3245-3310", rationale: "Background checks run at configurable interval"} -- {id: other.config.runtime_update.applies_at_runtime, module: other, tier: P0, area: config, assertions: [applies_at_runtime], source: "proxy_server.py:14014-14060", rationale: "/config/update persists to DB + invalidates cache"} -- {id: other.config.general_settings.alert_webhook_side_effect, module: other, tier: P1, area: config, assertions: [alert_webhook_side_effect], source: "proxy_server.py:14215", rationale: "alert_to_webhook_url auto-enables slack alerting"} -- {id: other.config.secret_resolution.kms_integration, module: other, tier: P1, area: config, assertions: [kms_integration], source: "proxy_server.py:3984-4010", rationale: "Resolves secrets from Vault/KMS at startup"} -- {id: other.config.overrides.audit_logged, module: other, tier: P1, area: config, assertions: [audit_logged], source: "config_override_endpoints.py:67-100", rationale: "Config override mutations audit-logged, values redacted"} -- {id: other.key_mgmt.regenerate.grace_period_honored, module: other, tier: P1, area: auth, assertions: [grace_period_honored], source: "key_management_endpoints.py:4503-4560", rationale: "Old key valid during grace_period then revoked"} -- {id: other.key_mgmt.spend_reset.resets_to_value, module: other, tier: P1, area: auth, assertions: [resets_to_value], source: "key_management_endpoints.py:4841", rationale: "reset_spend resets accumulated spend"} diff --git a/tests/e2e/coverage_registry/registry.py b/tests/e2e/coverage_registry/registry.py deleted file mode 100644 index 7a472cfbf2f..00000000000 --- a/tests/e2e/coverage_registry/registry.py +++ /dev/null @@ -1,26 +0,0 @@ -"""Load and validate the registry: the denominator, built in one shot from the YAMLs.""" - -from __future__ import annotations - -from collections import Counter -from pathlib import Path - -import yaml - -from .schema import CELL_ADAPTER, Cell - -REGISTRY_DIR = Path(__file__).resolve().parent - - -def load_registry(registry_dir: Path = REGISTRY_DIR) -> tuple[Cell, ...]: - """Every cell across every `*.yaml`, validated. Raises on a schema violation or - a duplicate id, since either would corrupt the coverage denominator.""" - cells = tuple( - CELL_ADAPTER.validate_python(row) - for path in sorted(registry_dir.glob("*.yaml")) - for row in (yaml.safe_load(path.read_text()) or ()) - ) - duplicates = sorted(cid for cid, n in Counter(c.id for c in cells).items() if n > 1) - if duplicates: - raise ValueError(f"duplicate cell ids in registry: {duplicates}") - return cells diff --git a/tests/e2e/coverage_registry/reliability.yaml b/tests/e2e/coverage_registry/reliability.yaml deleted file mode 100644 index ad5630a32dd..00000000000 --- a/tests/e2e/coverage_registry/reliability.yaml +++ /dev/null @@ -1,30 +0,0 @@ -# Reliability & Performance (behavior features). Grounded in litellm/router.py + router_strategy/ + router_utils/. -- {id: reliability.fallback.5xx.routes_to_fallback, module: reliability, tier: P0, behavior: fallback, variant: "5xx", assertions: [routes_to_fallback], exercised_on: [chat_completions, messages], source: "litellm/router.py:2024", rationale: "Reroute on provider 5xx to alternate deployment"} -- {id: reliability.fallback.context_window.routes_to_fallback, module: reliability, tier: P0, behavior: fallback, variant: context_window, assertions: [routes_to_fallback], exercised_on: [chat_completions, messages], source: "litellm/router.py:6108", rationale: "Fallback when model exceeds context limit"} -- {id: reliability.fallback.content_policy.routes_to_fallback, module: reliability, tier: P0, behavior: fallback, variant: content_policy, assertions: [routes_to_fallback], exercised_on: [chat_completions, messages], source: "litellm/router.py:6023", rationale: "Reroute on content-policy violation"} -- {id: reliability.fallback.timeout.routes_to_fallback, module: reliability, tier: P0, behavior: fallback, variant: "timeout", assertions: [routes_to_fallback], exercised_on: [chat_completions, messages], source: "litellm/router.py:2766", rationale: "Fallback on request timeout"} -- {id: reliability.retry.5xx.succeeds_within_retries, module: reliability, tier: P0, behavior: retry, variant: "5xx", assertions: [succeeds_within_retries], exercised_on: [chat_completions, messages], source: "litellm/router.py:6414", rationale: "Transient 5xx often succeeds on retry"} -- {id: reliability.retry.timeout.succeeds_within_retries, module: reliability, tier: P0, behavior: retry, variant: timeout, assertions: [succeeds_within_retries], exercised_on: [chat_completions, messages], source: "get_retry_from_policy.py:44", rationale: "Timeout retried per policy"} -- {id: reliability.retry.429.succeeds_within_retries, module: reliability, tier: P0, behavior: retry, variant: "429", assertions: [succeeds_within_retries], exercised_on: [chat_completions, messages], source: "get_retry_from_policy.py:46", rationale: "429 retried per RateLimitErrorRetries policy"} -- {id: reliability.retry.auth.succeeds_within_retries, module: reliability, tier: P1, behavior: retry, variant: auth, assertions: [succeeds_within_retries], exercised_on: [chat_completions, messages], source: "get_retry_from_policy.py:42", rationale: "Transient auth glitch retry"} -- {id: reliability.retry.context_window.succeeds_within_retries, module: reliability, tier: P1, behavior: retry, variant: context_window, assertions: [succeeds_within_retries], exercised_on: [chat_completions, messages], source: "get_retry_from_policy.py:51", rationale: "Multi-attempt on context error"} -- {id: reliability.cooldown.5xx.trips_then_recovers, module: reliability, tier: P0, behavior: cooldown, variant: "5xx", assertions: [trips_then_recovers], exercised_on: [chat_completions, messages], source: "cooldown_handlers.py:40", rationale: "Deployment cools after repeated 5xx, recovers after cooldown_time"} -- {id: reliability.cooldown.429.trips_then_recovers, module: reliability, tier: P0, behavior: cooldown, variant: "429", assertions: [trips_then_recovers], exercised_on: [chat_completions, messages], source: "cooldown_handlers.py:69", rationale: "Cools on 429, avoids hammering exhausted provider"} -- {id: reliability.cooldown.auth.trips_then_recovers, module: reliability, tier: P1, behavior: cooldown, variant: auth, assertions: [trips_then_recovers], exercised_on: [chat_completions, messages], source: "cooldown_handlers.py:74", rationale: "Cools on 401 auth error"} -- {id: reliability.cooldown.timeout.trips_then_recovers, module: reliability, tier: P1, behavior: cooldown, variant: timeout, assertions: [trips_then_recovers], exercised_on: [chat_completions, messages], source: "cooldown_handlers.py:77", rationale: "Cools on 408 timeout"} -- {id: reliability.ratelimit.rpm.blocks_over_limit, module: reliability, tier: P0, behavior: ratelimit, variant: rpm, assertions: [blocks_over_limit], exercised_on: [chat_completions, messages], source: "dynamic_rate_limiter_v3.py", rationale: "v3 limiter enforces RPM per key/team/model; 429 on breach"} -- {id: reliability.ratelimit.tpm.blocks_over_limit, module: reliability, tier: P0, behavior: ratelimit, variant: tpm, assertions: [blocks_over_limit], exercised_on: [chat_completions, messages], source: "dynamic_rate_limiter_v3.py", rationale: "v3 limiter enforces TPM per key/team/model; 429 on breach"} -- {id: reliability.ratelimit.priority_generous.picks_under_tpm, module: reliability, tier: P1, behavior: ratelimit, variant: priority_generous, assertions: [picks_under_tpm], exercised_on: [chat_completions, messages], source: "dynamic_rate_limiter_v3.py:36-52", rationale: "Generous mode (<80% sat) allows priority borrowing"} -- {id: reliability.ratelimit.priority_strict.picks_under_tpm, module: reliability, tier: P1, behavior: ratelimit, variant: priority_strict, assertions: [picks_under_tpm], exercised_on: [chat_completions, messages], source: "dynamic_rate_limiter_v3.py:53-71", rationale: "Strict mode (>=80% sat) enforces priority fairness"} -- {id: reliability.routing.simple_shuffle.picks_healthy_deployment, module: reliability, tier: P1, behavior: routing, variant: simple_shuffle, assertions: [picks_healthy_deployment], exercised_on: [chat_completions, messages], source: "router_strategy/simple_shuffle.py", rationale: "Baseline weighted/uniform pick"} -- {id: reliability.routing.latency_based.picks_lowest_latency, module: reliability, tier: P1, behavior: routing, variant: latency_based, assertions: [picks_lowest_latency], exercised_on: [chat_completions, messages], source: "router_strategy/lowest_latency.py", rationale: "Routes to lowest-latency deployment"} -- {id: reliability.routing.cost_based.picks_lowest_cost, module: reliability, tier: P1, behavior: routing, variant: cost_based, assertions: [picks_lowest_cost], exercised_on: [chat_completions, messages], source: "router_strategy/lowest_cost.py", rationale: "Spend-aware routing"} -- {id: reliability.routing.usage_based.picks_under_tpm, module: reliability, tier: P0, behavior: routing, variant: usage_based, assertions: [picks_under_tpm], exercised_on: [chat_completions, messages], source: "router_strategy/lowest_tpm_rpm_v2.py", rationale: "Routes to lowest-TPM deployment; prevents over-allocation"} -- {id: reliability.routing.least_busy.picks_lowest_traffic, module: reliability, tier: P1, behavior: routing, variant: least_busy, assertions: [picks_lowest_traffic], exercised_on: [chat_completions, messages], source: "router_strategy/least_busy.py", rationale: "Fewest in-flight requests"} -- {id: reliability.cache.exact.returns_cached, module: reliability, tier: P1, behavior: cache, variant: exact, assertions: [returns_cached], exercised_on: [chat_completions, messages, embeddings], source: "litellm/caching/caching.py", rationale: "Response cache returns cached on exact match"} -- {id: reliability.cache.prompt_caching_model_select.returns_cached, module: reliability, tier: P1, behavior: cache, variant: prompt_caching_model_select, assertions: [returns_cached], exercised_on: [chat_completions], source: "router_utils/prompt_caching_cache.py", rationale: "Selects model supporting prompt caching for cacheable prefix"} -- {id: reliability.circuit_breaker.redis.trips_then_recovers, module: reliability, tier: P0, behavior: circuit_breaker, variant: redis, assertions: [trips_then_recovers], exercised_on: [chat_completions, messages, embeddings], source: "litellm/caching/redis_cache.py:99", rationale: "Redis breaker CLOSED->OPEN->HALF_OPEN; guards all cache/rate-limit ops"} -- {id: reliability.timeout.request_timeout.exceeds_deadline, module: reliability, tier: P1, behavior: timeout, variant: request_timeout, assertions: [exceeds_deadline], exercised_on: [chat_completions, messages], source: "litellm/router.py:545-551", rationale: "Per-request timeout raises Timeout"} -- {id: reliability.timeout.stream_timeout.exceeds_deadline, module: reliability, tier: P1, behavior: timeout, variant: stream_timeout, assertions: [exceeds_deadline], exercised_on: [chat_completions], source: "litellm/router.py:551", rationale: "Streaming chunk-delivery timeout"} -- {id: reliability.perf.latency.under_slo, module: reliability, tier: P1, behavior: perf, variant: latency, assertions: [under_slo], exercised_on: [chat_completions, messages], source: "router_strategy/lowest_latency.py", rationale: "Latency SLO (p50/p99) compliance"} -- {id: reliability.perf.throughput.under_slo, module: reliability, tier: P1, behavior: perf, variant: throughput, assertions: [under_slo], exercised_on: [chat_completions, messages], source: grammar, rationale: "Throughput SLO under load"} diff --git a/tests/e2e/coverage_registry/schema.py b/tests/e2e/coverage_registry/schema.py index bb27fbf0ea0..4fc819fe704 100644 --- a/tests/e2e/coverage_registry/schema.py +++ b/tests/e2e/coverage_registry/schema.py @@ -1,182 +1,180 @@ -"""Registry row schema: the contract every denominator cell validates against. +"""Schema for e2e coverage declared directly on pytest tests. -A cell is one customer-noticeable behavior a single e2e test can assert pass/fail -on. `module` is the id's segment-1 prefix (seven of them); dashboard rollups can -split or merge those prefixes. The union is discriminated on `module`, so an LLM -row cannot carry a guardrail field and vice versa. +Each collected e2e pytest must provide: + + @pytest.mark.e2e_coverage( + module="core_llms", + endpoint="/chat/completions", + provider="openai", + params=["tools"], + ) + +The collector turns those markers into endpoint x provider x parameter coverage +units and Grafana-ready module summaries. """ from __future__ import annotations -from enum import Enum -from typing import Annotated, Literal +import re +from typing import Literal -from pydantic import BaseModel, ConfigDict, Field, TypeAdapter +from pydantic import BaseModel, ConfigDict, Field, field_validator - -class Tier(str, Enum): - P0 = "P0" - P1 = "P1" - P2 = "P2" - - -class FailBeforeFix(str, Enum): - proven = "proven" - unproven = "unproven" - - -LlmEndpoint = Literal[ - "chat_completions", - "messages", - "responses", - "embeddings", - "batches", - "files", - "rerank", - "images_generations", - "audio_speech", - "audio_transcriptions", - "moderations", - "realtime", +CoverageModule = Literal[ + "core_llms", + "non_core_llms", + "access_control", + "budgets", + "spend_tracking", + "management", + "mcp", + "rate_limits", + "reliability", + "logging", + "guardrails", + "other", ] -LlmRoute = Literal[ - "anthropic", - "azure_foundry", - "azure_openai", - "bedrock_converse", - "cohere", - "openai", - "together_ai", - "vertex", -] - -LlmCapability = Literal[ - "basic", - "prompt_cache_5m", - "service_tier", - "structured_output", - "thinking", - "tool_use", - "vision", -] - - -class _Base(BaseModel): - model_config = ConfigDict(frozen=True, extra="forbid") - - id: str - tier: Tier - assertions: tuple[str, ...] - source: str - rationale: str = "" - fail_before_fix: FailBeforeFix = FailBeforeFix.unproven - supported: bool = True - - -class LlmCell(_Base): - module: Literal["llm"] - subject_endpoint: LlmEndpoint - route: LlmRoute - capability: LlmCapability - streaming: Literal["stream", "nonstream", "na"] - - -class MgmtCell(_Base): - module: Literal["mgmt"] - surface: Literal["api", "ui"] - - -class McpCell(_Base): - module: Literal["mcp"] - operation: str - auth_family: Literal["none", "api_key", "bearer", "oauth"] - - -class ReliabilityCell(_Base): - module: Literal["reliability"] - behavior: str - variant: str - exercised_on: tuple[str, ...] - - -class LoggingCell(_Base): - module: Literal["logging"] - event: str - exercised_on: tuple[str, ...] - - -class GuardrailCell(_Base): - module: Literal["guardrail"] - hook_point: str - exercised_on: tuple[str, ...] - - -class OtherCell(_Base): - module: Literal["other"] - area: str - - -Cell = Annotated[ - LlmCell - | MgmtCell - | McpCell - | ReliabilityCell - | LoggingCell - | GuardrailCell - | OtherCell, - Field(discriminator="module"), -] - -CELL_ADAPTER: TypeAdapter[Cell] = TypeAdapter(Cell) - -CORE_LLM_ENDPOINTS: frozenset[str] = frozenset( - { - "chat_completions", - "messages", - "responses", - } +MODULE_ORDER: tuple[CoverageModule, ...] = ( + "core_llms", + "non_core_llms", + "access_control", + "budgets", + "spend_tracking", + "management", + "mcp", + "rate_limits", + "reliability", + "logging", + "guardrails", + "other", ) -PREFIX_ROLLUP: dict[str, str] = { - "mcp": "MCPs", - "mgmt": "Management/UI", - "reliability": "Reliability & Performance", - "logging": "Logging & Guardrails", - "guardrail": "Logging & Guardrails", +MODULE_DISPLAY_NAMES: dict[str, str] = { + "core_llms": "Core LLMs", + "non_core_llms": "Non-Core LLMs", + "access_control": "Access Control", + "budgets": "Budgets", + "spend_tracking": "Spend Tracking", + "management": "Management", + "mcp": "MCP", + "rate_limits": "Rate Limits", + "reliability": "Reliability", + "logging": "Logging", + "guardrails": "Guardrails", "other": "Other", } -MODULE_ORDER: tuple[str, ...] = ( - "Core LLMs", - "Non-Core LLMs", - "MCPs", - "Management/UI", - "Reliability & Performance", - "Logging & Guardrails", - "Other", +KNOWN_ENDPOINTS: frozenset[str] = frozenset( + { + "/chat/completions", + "/v1/messages", + "/v1/responses", + "/v1/batches", + "/v1/realtime", + "/v1/audio/speech", + "/v1/embeddings", + "/v1/images/generations", + "/rerank", + "/anthropic/*", + "/vertex_ai/*", + "/model/*", + "/key/*", + "/team/*", + "/user/*", + "/organization/*", + "/budget/*", + "/spend/*", + "/global/spend/*", + "/mcp/*", + "/guardrails/*", + "/health/*", + "coverage_registry", + "e2e_harness", + "logging", + "reliability", + } ) -LOKI_MODULE_LABELS: dict[str, str] = { - "Core LLMs": "core_llms", - "Non-Core LLMs": "non_core_llms", - "MCPs": "mcp", - "Management/UI": "management_ui", - "Reliability & Performance": "reliability_performance", - "Logging & Guardrails": "logging_guardrails", - "Other": "other", -} +KNOWN_PROVIDERS: frozenset[str] = frozenset( + { + "anthropic", + "azure", + "azure_ai", + "azure_document_intelligence", + "bedrock", + "cohere", + "deepseek", + "gemini", + "litellm", + "mistral", + "multiple", + "openai", + "prometheus", + "proxy", + "vertex_ai", + "xai", + } +) + +_PARAM_RE = re.compile(r"^[a-z0-9][a-z0-9_.:/-]*$") -def dashboard_module(cell: Cell) -> str: - """Return the Grafana/reporting module for a registry cell.""" - if isinstance(cell, LlmCell): - if cell.subject_endpoint in CORE_LLM_ENDPOINTS: - return "Core LLMs" - return "Non-Core LLMs" - return PREFIX_ROLLUP[cell.module] +class CoveragePoint(BaseModel): + """Validated data carried by @pytest.mark.e2e_coverage.""" + + model_config = ConfigDict(frozen=True, extra="forbid") + + module: CoverageModule + endpoint: str + provider: str + params: tuple[str, ...] = Field(min_length=1) + + @field_validator("endpoint") + @classmethod + def endpoint_must_be_known(cls, value: str) -> str: + if value not in KNOWN_ENDPOINTS: + raise ValueError(f"unknown endpoint {value!r}") + return value + + @field_validator("provider") + @classmethod + def provider_must_be_known(cls, value: str) -> str: + if value not in KNOWN_PROVIDERS: + raise ValueError(f"unknown provider {value!r}") + return value + + @field_validator("params") + @classmethod + def params_must_be_normalized(cls, value: tuple[str, ...]) -> tuple[str, ...]: + invalid = [param for param in value if not _PARAM_RE.fullmatch(param)] + if invalid: + raise ValueError(f"invalid params: {invalid}") + return value -def loki_module_label(module: str) -> str: - """Return the log-safe Loki label for a dashboard module.""" - return LOKI_MODULE_LABELS[module] +class CoverageUnit(BaseModel): + """One endpoint x provider x parameter combination covered by a test.""" + + model_config = ConfigDict(frozen=True) + + module: CoverageModule + endpoint: str + provider: str + param: str + + @property + def key(self) -> str: + return f"{self.module}|{self.endpoint}|{self.provider}|{self.param}" + + +def units_for_point(point: CoveragePoint) -> tuple[CoverageUnit, ...]: + return tuple( + CoverageUnit( + module=point.module, + endpoint=point.endpoint, + provider=point.provider, + param=param, + ) + for param in point.params + ) diff --git a/tests/e2e/coverage_registry/test_collector.py b/tests/e2e/coverage_registry/test_collector.py index 079ee215866..d5cdc95fffa 100644 --- a/tests/e2e/coverage_registry/test_collector.py +++ b/tests/e2e/coverage_registry/test_collector.py @@ -1,194 +1,124 @@ -"""Tests for the coverage-registry tooling: pure logic plus a registry canary. - -No `e2e` marker, so these run without a proxy. They exercise the coverage math and -the registry loader, and guard the checked-in registry against schema drift and -duplicate ids. -""" +"""Tests for marker-based e2e coverage collection.""" from __future__ import annotations -from pathlib import Path - import pytest +from coverage_registry.check_coverage_sync import sync_errors from coverage_registry.collector import ( + CoverageMarkers, compute_coverage, - render, render_json, render_loki, render_prometheus, ) -from coverage_registry.registry import load_registry -from coverage_registry.schema import ( - GuardrailCell, - LlmCell, - LlmEndpoint, - LoggingCell, - Tier, - loki_module_label, +from coverage_registry.schema import CoveragePoint + +pytestmark = pytest.mark.e2e_coverage( + module="other", + endpoint="coverage_registry", + provider="proxy", + params=["collector", "sync_checker", "schema"], ) -def _llm( - cell_id: str, tier: Tier, subject_endpoint: LlmEndpoint = "chat_completions" -) -> LlmCell: - return LlmCell( - id=cell_id, - module="llm", - tier=tier, - assertions=("works",), - source="test", - subject_endpoint=subject_endpoint, - route="openai", - capability="basic", - streaming="nonstream", - ) - - -def test_compute_coverage_counts_covered_p0_and_gaps() -> None: - cells = (_llm("llm.a", Tier.P0), _llm("llm.b", Tier.P0), _llm("llm.c", Tier.P1)) - report = compute_coverage(cells, frozenset({"llm.a"})) - assert (report.total, report.covered) == (3, 1) - assert (report.p0_total, report.p0_covered) == (2, 1) - assert report.p0_gaps == ("llm.b",) - assert report.orphan_markers == () - - -def test_orphan_marker_is_reported_not_counted() -> None: - cells = (_llm("llm.a", Tier.P0),) - report = compute_coverage(cells, frozenset({"llm.a", "llm.ghost"})) - assert report.covered == 1 - assert report.orphan_markers == ("llm.ghost",) - - -def test_logging_and_guardrail_roll_up_into_one_module() -> None: - cells = ( - LoggingCell( - id="logging.x", - module="logging", - tier=Tier.P0, - assertions=("logs_spend",), - source="t", - event="success", - exercised_on=("chat_completions",), +def _markers() -> CoverageMarkers: + return CoverageMarkers( + points_by_nodeid={ + "test_core.py::test_chat": ( + CoveragePoint( + module="core_llms", + endpoint="/chat/completions", + provider="openai", + params=("tools", "streaming"), + ), + ), + "test_budget.py::test_budget": ( + CoveragePoint( + module="budgets", + endpoint="/chat/completions", + provider="proxy", + params=("budget_enforcement",), + ), + ), + }, + collected_nodeids=( + "test_budget.py::test_budget", + "test_core.py::test_chat", + "test_unmarked.py::test_missing", ), - GuardrailCell( - id="guardrail.y", - module="guardrail", - tier=Tier.P1, - assertions=("blocks",), - source="t", - hook_point="pre_call", - exercised_on=("chat_completions",), - ), - ) - report = compute_coverage(cells, frozenset()) - logging_and_guardrails = next( - m for m in report.modules if m.module == "Logging & Guardrails" - ) - assert logging_and_guardrails.total == 2 - - -def test_llm_cells_roll_up_by_core_endpoint() -> None: - cells = ( - _llm("llm.chat", Tier.P0, "chat_completions"), - _llm("llm.messages", Tier.P0, "messages"), - _llm("llm.responses", Tier.P1, "responses"), - _llm("llm.batches", Tier.P0, "batches"), - _llm("llm.realtime", Tier.P1, "realtime"), - ) - report = compute_coverage(cells, frozenset({"llm.chat", "llm.batches"})) - - core = next(m for m in report.modules if m.module == "Core LLMs") - non_core = next(m for m in report.modules if m.module == "Non-Core LLMs") - - assert (core.total, core.covered, core.p0_total, core.p0_covered) == (3, 1, 2, 1) - assert ( - non_core.total, - non_core.covered, - non_core.p0_total, - non_core.p0_covered, - ) == (2, 1, 1, 1) - - -def test_text_render_uses_plain_coverage_language() -> None: - report = compute_coverage( - (_llm("llm.chat", Tier.P0), _llm("llm.batches", Tier.P0, "batches")), - frozenset({"llm.chat"}), + unmarked_nodeids=("test_unmarked.py::test_missing",), + invalid_markers=(), + collection_errors=(), ) - text = render(report) - assert "COVERAGE" in text - assert "Headline coverage: 1/2 (50.0%)" in text - assert "P0 COVERED" not in text +def test_compute_coverage_counts_unique_units_and_tests() -> None: + report = compute_coverage(_markers()) + + core = next(m for m in report.modules if m.module == "core_llms") + budgets = next(m for m in report.modules if m.module == "budgets") + + assert report.total == 3 + assert report.test_count == 2 + assert report.collected_test_count == 3 + assert report.unmarked_test_count == 1 + assert (core.unit_count, core.test_count) == (2, 1) + assert (budgets.unit_count, budgets.test_count) == (1, 1) -def test_json_render_exposes_module_coverage_for_grafana_jobs() -> None: - report = compute_coverage( - (_llm("llm.chat", Tier.P0), _llm("llm.batches", Tier.P0, "batches")), - frozenset({"llm.chat"}), - ) +def test_json_render_exposes_marker_fields_for_grafana_jobs() -> None: + payload = render_json(compute_coverage(_markers())) - payload = render_json(report) - - assert '"coverage_percent": 50.0' in payload - assert '"module": "Core LLMs"' in payload - assert '"module": "Non-Core LLMs"' in payload + assert '"module": "core_llms"' in payload + assert '"endpoint": "/chat/completions"' in payload + assert '"provider": "openai"' in payload + assert '"param": "tools"' in payload + assert '"unmarked_test_count": 1' in payload -def test_prometheus_render_exposes_module_coverage_timeseries() -> None: - report = compute_coverage( - (_llm("llm.chat", Tier.P0), _llm("llm.batches", Tier.P0, "batches")), - frozenset({"llm.chat"}), - ) +def test_prometheus_render_exposes_module_counts() -> None: + metrics = render_prometheus(compute_coverage(_markers())) - metrics = render_prometheus(report) - - assert 'litellm_e2e_coverage_cells{module="Core LLMs",state="covered"} 1' in metrics - assert 'litellm_e2e_coverage_percent{module="Core LLMs"} 100.000000' in metrics - assert 'litellm_e2e_coverage_percent{module="Non-Core LLMs"} 0.000000' in metrics - assert "litellm_e2e_coverage_orphan_markers 0" in metrics + assert 'litellm_e2e_coverage_units{module="core_llms"} 2' in metrics + assert 'litellm_e2e_coverage_tests{module="budgets"} 1' in metrics + assert "litellm_e2e_coverage_unmarked_tests 1" in metrics + assert "litellm_e2e_coverage_invalid_markers 0" in metrics def test_loki_render_exposes_exact_stdout_lines_for_loki() -> None: - report = compute_coverage( - (_llm("llm.chat", Tier.P0), _llm("llm.batches", Tier.P0, "batches")), - frozenset({"llm.chat"}), - ) - + report = compute_coverage(_markers()) lines = render_loki(report).splitlines() assert len(lines) == 1 + len(report.modules) - assert lines[0] == "COVERAGE_TOTAL percent=50.0 covered=1 total=2" assert ( - lines[1] == "COVERAGE_MODULE module=core_llms percent=100.0 covered=1 total=1" + lines[0] == "COVERAGE_TOTAL percent=100.0 covered=3 total=3 tests=2 " + "unmarked_tests=1 invalid_markers=0" ) - assert ( - lines[2] == "COVERAGE_MODULE module=non_core_llms percent=0.0 covered=0 total=1" - ) - assert [line.split("module=", 1)[1].split(" ", 1)[0] for line in lines[1:]] == [ - loki_module_label(module.module) for module in report.modules - ] - assert all( - " " not in line.split("module=", 1)[1].split(" ", 1)[0] for line in lines[1:] + assert lines[1] == ( + "COVERAGE_MODULE module=core_llms percent=100.0 covered=2 total=2 tests=1" ) -def test_real_registry_loads_and_ids_are_unique() -> None: - cells = load_registry() - ids = [c.id for c in cells] - assert len(cells) > 250 - assert len(ids) == len(set(ids)) - assert any(c.id == "logging.prometheus.success.exports_metric" for c in cells) +def test_marker_schema_rejects_unknown_endpoint() -> None: + with pytest.raises(ValueError, match="unknown endpoint"): + CoveragePoint( + module="core_llms", + endpoint="/not-real", + provider="openai", + params=("tools",), + ) -def test_load_registry_rejects_duplicate_ids(tmp_path: Path) -> None: - row = ( - "- {id: llm.dup, module: llm, tier: P0, assertions: [works], source: t, " - "subject_endpoint: chat_completions, route: openai, capability: basic, streaming: nonstream}\n" - ) - (tmp_path / "a.yaml").write_text(row) - (tmp_path / "b.yaml").write_text(row) - with pytest.raises(ValueError, match="duplicate cell ids"): - load_registry(tmp_path) +def test_marker_schema_requires_params() -> None: + with pytest.raises(ValueError, match="at least 1 item"): + CoveragePoint( + module="core_llms", + endpoint="/chat/completions", + provider="openai", + params=(), + ) + + +def test_collected_e2e_tests_have_coverage_metadata() -> None: + assert sync_errors() == () diff --git a/tests/e2e/batches/COVERAGE.md b/tests/e2e/llm_translation/batches/COVERAGE.md similarity index 100% rename from tests/e2e/batches/COVERAGE.md rename to tests/e2e/llm_translation/batches/COVERAGE.md diff --git a/tests/e2e/batches/batch_client.py b/tests/e2e/llm_translation/batches/batch_client.py similarity index 100% rename from tests/e2e/batches/batch_client.py rename to tests/e2e/llm_translation/batches/batch_client.py diff --git a/tests/e2e/batches/capabilities.py b/tests/e2e/llm_translation/batches/capabilities.py similarity index 98% rename from tests/e2e/batches/capabilities.py rename to tests/e2e/llm_translation/batches/capabilities.py index 522e3162e24..e19aeb45155 100644 --- a/tests/e2e/batches/capabilities.py +++ b/tests/e2e/llm_translation/batches/capabilities.py @@ -97,7 +97,9 @@ class Capability: PROVIDERS: tuple[Provider, ...] = ( Provider("openai", "openai-batch", "gpt-4o-mini", can_cancel=True, can_list=True), - Provider("azure", "azure-batch", "gpt-4.1-mini-batch", can_cancel=True, can_list=True), + Provider( + "azure", "azure-batch", "gpt-4.1-mini-batch", can_cancel=True, can_list=True + ), Provider( "vertex_ai", "vertex-batch", "gemini-2.5-flash", can_cancel=True, can_list=True ), diff --git a/tests/e2e/batches/conftest.py b/tests/e2e/llm_translation/batches/conftest.py similarity index 91% rename from tests/e2e/batches/conftest.py rename to tests/e2e/llm_translation/batches/conftest.py index 2c6070c437a..60831503af7 100644 --- a/tests/e2e/batches/conftest.py +++ b/tests/e2e/llm_translation/batches/conftest.py @@ -16,9 +16,9 @@ from typing import Iterator import pytest -from batch_client import BatchClient, build_client -from capabilities import PROVIDERS from e2e_http import NoBody +from llm_translation.batches.batch_client import BatchClient, build_client +from llm_translation.batches.capabilities import PROVIDERS def pytest_configure(config: pytest.Config) -> None: diff --git a/tests/e2e/batches/test_batches_e2e.py b/tests/e2e/llm_translation/batches/test_batches_e2e.py similarity index 94% rename from tests/e2e/batches/test_batches_e2e.py rename to tests/e2e/llm_translation/batches/test_batches_e2e.py index a998f962c04..2c08c95fe26 100644 --- a/tests/e2e/batches/test_batches_e2e.py +++ b/tests/e2e/llm_translation/batches/test_batches_e2e.py @@ -23,7 +23,7 @@ import pytest from e2e_config import unique_marker -from batch_client import ( +from llm_translation.batches.batch_client import ( BatchClient, BatchCreateBody, BatchObject, @@ -31,7 +31,7 @@ from batch_client import ( is_model_access_denied, is_result_access_denied, ) -from capabilities import ( +from llm_translation.batches.capabilities import ( BATCH_ID_SHAPE, CAPABILITIES, FILE_ID_SHAPE, @@ -51,7 +51,20 @@ from e2e_http import ( from lifecycle import ResourceManager from models import KeyGenerateBody, SpendLogRow, SpendLogsParams -pytestmark = pytest.mark.e2e +pytestmark = [ + pytest.mark.e2e, + pytest.mark.e2e_coverage( + module="non_core_llms", + endpoint="/v1/batches", + provider="multiple", + params=[ + "batch_lifecycle", + "file_upload", + "key_model_access", + "rate_limit_spend", + ], + ), +] CREATED_BATCH_STATUSES = {"validating", "in_progress", "finalizing"} BATCH_CANCEL_DELAY_SECONDS = 2 @@ -215,9 +228,7 @@ def test_batch_lifecycle( if cap.can_cancel: time.sleep(BATCH_CANCEL_DELAY_SECONDS) pre_cancel = unwrap(client.retrieve_batch(batch.id, key=key, provider=provider)) - assert ( - pre_cancel.status not in BATCH_TERMINAL_BEFORE_CANCEL - ), ( + assert pre_cancel.status not in BATCH_TERMINAL_BEFORE_CANCEL, ( f"batch reached {pre_cancel.status!r} before cancel; " "provider likely rejected the input" ) @@ -229,9 +240,9 @@ def test_batch_lifecycle( valid_post_cancel = {"cancelling", "cancelled"} if cap.provider == "vertex_ai": valid_post_cancel |= CREATED_BATCH_STATUSES - assert cancelled.status in valid_post_cancel, ( - f"unexpected post-cancel status {cancelled.status!r}" - ) + assert ( + cancelled.status in valid_post_cancel + ), f"unexpected post-cancel status {cancelled.status!r}" if cap.can_list: listed = unwrap(client.list_batches(key=key, provider=provider)) @@ -329,12 +340,15 @@ def test_rate_limited_batch_create_leaves_no_unattributed_spend_row( """ user_id = f"e2e-batch-rl-{unique_marker()}" key = client.gateway.generate_key( - KeyGenerateBody(models=[], tpm_limit=1_000_000, rpm_limit=1_000, user_id=user_id) + KeyGenerateBody( + models=[], tpm_limit=1_000_000, rpm_limit=1_000, user_id=user_id + ) ) resources.defer(lambda: client.gateway.delete_key(key)) before = frozenset( - row.request_id for row in unattributed_rows(client.gateway.spend_logs(SpendLogsParams())) + row.request_id + for row in unattributed_rows(client.gateway.spend_logs(SpendLogsParams())) ) file = unwrap( diff --git a/tests/e2e/realtime/REALTIME_COVERAGE_MATRIX.md b/tests/e2e/llm_translation/realtime/REALTIME_COVERAGE_MATRIX.md similarity index 97% rename from tests/e2e/realtime/REALTIME_COVERAGE_MATRIX.md rename to tests/e2e/llm_translation/realtime/REALTIME_COVERAGE_MATRIX.md index 8624475d0de..0bc21d90621 100644 --- a/tests/e2e/realtime/REALTIME_COVERAGE_MATRIX.md +++ b/tests/e2e/llm_translation/realtime/REALTIME_COVERAGE_MATRIX.md @@ -48,7 +48,7 @@ Start a proxy with the gateway config and the provider keys set in its environment, then ``` -uv run pytest tests/e2e/realtime/ -v +uv run pytest tests/e2e/llm_translation/realtime/ -v ``` Tests skip when no proxy answers `GET /health/liveliness` at `LITELLM_PROXY_URL` diff --git a/tests/e2e/realtime/conftest.py b/tests/e2e/llm_translation/realtime/conftest.py similarity index 86% rename from tests/e2e/realtime/conftest.py rename to tests/e2e/llm_translation/realtime/conftest.py index 4a5c4837a1a..5d777c7ade9 100644 --- a/tests/e2e/realtime/conftest.py +++ b/tests/e2e/llm_translation/realtime/conftest.py @@ -7,7 +7,7 @@ Gateway, so the `resources` fixture cleans up keys this suite creates. import pytest -from realtime_client import RealtimeClient, build_client +from llm_translation.realtime.realtime_client import RealtimeClient, build_client @pytest.fixture(scope="session") diff --git a/tests/e2e/realtime/fixtures/weather_question_24k.wav b/tests/e2e/llm_translation/realtime/fixtures/weather_question_24k.wav similarity index 100% rename from tests/e2e/realtime/fixtures/weather_question_24k.wav rename to tests/e2e/llm_translation/realtime/fixtures/weather_question_24k.wav diff --git a/tests/e2e/realtime/pipecat_service.py b/tests/e2e/llm_translation/realtime/pipecat_service.py similarity index 100% rename from tests/e2e/realtime/pipecat_service.py rename to tests/e2e/llm_translation/realtime/pipecat_service.py diff --git a/tests/e2e/realtime/realtime_client.py b/tests/e2e/llm_translation/realtime/realtime_client.py similarity index 100% rename from tests/e2e/realtime/realtime_client.py rename to tests/e2e/llm_translation/realtime/realtime_client.py diff --git a/tests/e2e/realtime/test_realtime_e2e.py b/tests/e2e/llm_translation/realtime/test_realtime_e2e.py similarity index 94% rename from tests/e2e/realtime/test_realtime_e2e.py rename to tests/e2e/llm_translation/realtime/test_realtime_e2e.py index 01356900141..d0cb5c8ab55 100644 --- a/tests/e2e/realtime/test_realtime_e2e.py +++ b/tests/e2e/llm_translation/realtime/test_realtime_e2e.py @@ -15,7 +15,7 @@ REALTIME_COVERAGE_MATRIX.md. import pytest from pydantic import BaseModel -from realtime_client import ( +from llm_translation.realtime.realtime_client import ( PROVIDERS, ConversationItemCreate, FunctionCallArgumentsDone, @@ -36,7 +36,15 @@ from realtime_client import ( user_message, ) -pytestmark = pytest.mark.e2e +pytestmark = [ + pytest.mark.e2e, + pytest.mark.e2e_coverage( + module="non_core_llms", + endpoint="/v1/realtime", + provider="multiple", + params=["text_conversation", "tool_calling"], + ), +] PROVIDER_PARAMS = [pytest.param(p, id=p.id) for p in PROVIDERS] diff --git a/tests/e2e/realtime/test_realtime_pipecat_audio_e2e.py b/tests/e2e/llm_translation/realtime/test_realtime_pipecat_audio_e2e.py similarity index 97% rename from tests/e2e/realtime/test_realtime_pipecat_audio_e2e.py rename to tests/e2e/llm_translation/realtime/test_realtime_pipecat_audio_e2e.py index d9d7c744f66..c6ee8c497f0 100644 --- a/tests/e2e/realtime/test_realtime_pipecat_audio_e2e.py +++ b/tests/e2e/llm_translation/realtime/test_realtime_pipecat_audio_e2e.py @@ -27,14 +27,22 @@ from pathlib import Path import pytest -from realtime_client import ( +from llm_translation.realtime.realtime_client import ( PROVIDERS, RealtimeProvider, _ws_base_url, skip_if_unconfigured, ) -pytestmark = pytest.mark.e2e +pytestmark = [ + pytest.mark.e2e, + pytest.mark.e2e_coverage( + module="non_core_llms", + endpoint="/v1/realtime", + provider="multiple", + params=["audio_session"], + ), +] pytest.importorskip("pipecat", reason="pipecat-ai not installed") @@ -64,7 +72,9 @@ from pipecat.services.llm_service import FunctionCallParams # noqa: E402 from pipecat.services.openai.realtime import events as rt_events # noqa: E402 from pipecat.services.openai.realtime.llm import OpenAIRealtimeLLMService # noqa: E402 -from pipecat_service import LiteLLMRealtimeLLMService # noqa: E402 +from llm_translation.realtime.pipecat_service import ( # noqa: E402 + LiteLLMRealtimeLLMService, +) PROVIDER_PARAMS = [pytest.param(p, id=p.id) for p in PROVIDERS] diff --git a/tests/e2e/realtime/test_realtime_pipecat_e2e.py b/tests/e2e/llm_translation/realtime/test_realtime_pipecat_e2e.py similarity index 92% rename from tests/e2e/realtime/test_realtime_pipecat_e2e.py rename to tests/e2e/llm_translation/realtime/test_realtime_pipecat_e2e.py index 1068c54fdec..89c1cfdba9b 100644 --- a/tests/e2e/realtime/test_realtime_pipecat_e2e.py +++ b/tests/e2e/llm_translation/realtime/test_realtime_pipecat_e2e.py @@ -25,14 +25,22 @@ import asyncio import pytest -from realtime_client import ( +from llm_translation.realtime.realtime_client import ( PROVIDERS, RealtimeProvider, _ws_base_url, skip_if_unconfigured, ) -pytestmark = pytest.mark.e2e +pytestmark = [ + pytest.mark.e2e, + pytest.mark.e2e_coverage( + module="non_core_llms", + endpoint="/v1/realtime", + provider="multiple", + params=["pipecat_session"], + ), +] pytest.importorskip("pipecat", reason="pipecat-ai not installed") @@ -58,7 +66,9 @@ from pipecat.processors.frame_processor import ( # noqa: E402 ) from pipecat.services.llm_service import FunctionCallParams # noqa: E402 -from pipecat_service import LiteLLMRealtimeLLMService # noqa: E402 +from llm_translation.realtime.pipecat_service import ( # noqa: E402 + LiteLLMRealtimeLLMService, +) PROVIDER_PARAMS = [pytest.param(p, id=p.id) for p in PROVIDERS] diff --git a/tests/e2e/llm_translation/test_audio_speech_e2e.py b/tests/e2e/llm_translation/test_audio_speech_e2e.py index f7a04d94cb3..c03b9f16e8f 100644 --- a/tests/e2e/llm_translation/test_audio_speech_e2e.py +++ b/tests/e2e/llm_translation/test_audio_speech_e2e.py @@ -15,7 +15,15 @@ from endpoints_client import EndpointsClient from lifecycle import ResourceManager from models import LiteLLMParamsBody -pytestmark = pytest.mark.e2e +pytestmark = [ + pytest.mark.e2e, + pytest.mark.e2e_coverage( + module="non_core_llms", + endpoint="/v1/audio/speech", + provider="openai", + params=["audio_speech"], + ), +] class TestAudioSpeech: @@ -34,7 +42,7 @@ class TestAudioSpeech: result = endpoints_client.audio_speech(key, model, "Hello!") require_successful_call(result) - assert "audio" in (result.content_type or ""), ( - f"/audio/speech content-type is not audio: {result.content_type!r}" - ) + assert "audio" in ( + result.content_type or "" + ), f"/audio/speech content-type is not audio: {result.content_type!r}" assert result.body, "/audio/speech returned an empty body" diff --git a/tests/e2e/llm_translation/test_chat_completions_regression_e2e.py b/tests/e2e/llm_translation/test_chat_completions_regression_e2e.py index 5cc4ff308fa..3b559d8f651 100644 --- a/tests/e2e/llm_translation/test_chat_completions_regression_e2e.py +++ b/tests/e2e/llm_translation/test_chat_completions_regression_e2e.py @@ -18,7 +18,15 @@ from e2e_http import unwrap from models import ChatBody, ChatMessage from passthrough_client import PassthroughClient -pytestmark = pytest.mark.e2e +pytestmark = [ + pytest.mark.e2e, + pytest.mark.e2e_coverage( + module="core_llms", + endpoint="/chat/completions", + provider="multiple", + params=["chat_completions_regression"], + ), +] CHAT_MODELS: tuple[tuple[str, str], ...] = ( ("gpt-5.5", "openai"), @@ -33,12 +41,6 @@ class TestChatCompletionsRegression: CHAT_MODELS, ids=[f"{model}-{route}" for model, route in CHAT_MODELS], ) - @pytest.mark.covers( - "llm.chat_completions.openai.basic.nonstream.works", - "llm.chat_completions.anthropic.basic.nonstream.works", - "llm.chat_completions.vertex.basic.nonstream.works", - exercised_on=[], - ) def test_chat_returns_real_completion( self, client: PassthroughClient, scoped_key: str, model: str, route: str ) -> None: diff --git a/tests/e2e/llm_translation/test_custom_pricing_e2e.py b/tests/e2e/llm_translation/test_custom_pricing_e2e.py index 7894b447be9..c6bc22f6ac9 100644 --- a/tests/e2e/llm_translation/test_custom_pricing_e2e.py +++ b/tests/e2e/llm_translation/test_custom_pricing_e2e.py @@ -35,7 +35,15 @@ from models import ( SpendLogsParams, ) -pytestmark = pytest.mark.e2e +pytestmark = [ + pytest.mark.e2e, + pytest.mark.e2e_coverage( + module="core_llms", + endpoint="/chat/completions", + provider="proxy", + params=["custom_pricing", "model_info"], + ), +] BACKEND_MODEL = "gemini/gemini-2.5-flash" GEMINI_API_KEY = "os.environ/GEMINI_API_KEY" @@ -116,7 +124,9 @@ def _model_info_entry(entries: list[ModelInfoEntry], model_name: str) -> ModelIn pytest.fail(f"{model_name} absent from /model/info; the override did not load") -def _poll_breakdown_row(gateway: Gateway, key: str, response_id: str | None) -> _SpendRow: +def _poll_breakdown_row( + gateway: Gateway, key: str, response_id: str | None +) -> _SpendRow: """Poll /spend/logs until the call's row lands with a cost breakdown (rows flush ~60s behind the call via proxy_batch_write_at).""" deadline = time.monotonic() + gateway.poll_timeout @@ -164,7 +174,8 @@ class TestCustomPricing: model=model, messages=[ ChatMessage( - role="user", content=f"reply with one word {unique_marker()}" + role="user", + content=f"reply with one word {unique_marker()}", ) ], max_tokens=16, @@ -178,13 +189,15 @@ class TestCustomPricing: prompt = row.prompt_tokens or 0 completion = row.completion_tokens or 0 - assert prompt > 0 and completion > 0, f"call tokens not logged on the row: {row}" + assert ( + prompt > 0 and completion > 0 + ), f"call tokens not logged on the row: {row}" input_cost = breakdown.input_cost output_cost = breakdown.output_cost - assert input_cost is not None and output_cost is not None, ( - f"row cost breakdown missing input/output cost: {breakdown}" - ) + assert ( + input_cost is not None and output_cost is not None + ), f"row cost breakdown missing input/output cost: {breakdown}" assert _approx_equal(input_cost, prompt * CUSTOM_INPUT_RATE), ( f"input_cost {input_cost} != {prompt} tokens * {CUSTOM_INPUT_RATE} " f"= {prompt * CUSTOM_INPUT_RATE}" @@ -223,7 +236,9 @@ class TestCustomPricing: output_cost_per_token=None, ) - entries = {entry.model_name: entry for entry in endpoints_client.gateway.model_info()} + entries = { + entry.model_name: entry for entry in endpoints_client.gateway.model_info() + } custom_entry = entries.get(custom) sibling_entry = entries.get(sibling) assert custom_entry is not None, f"{custom} absent from /model/info" diff --git a/tests/e2e/llm_translation/test_deepseek_reasoning_e2e.py b/tests/e2e/llm_translation/test_deepseek_reasoning_e2e.py index f8f229aa2a7..7dcf04ad064 100644 --- a/tests/e2e/llm_translation/test_deepseek_reasoning_e2e.py +++ b/tests/e2e/llm_translation/test_deepseek_reasoning_e2e.py @@ -28,7 +28,15 @@ from lifecycle import ResourceManager from models import ChatBody, ChatMessage, ChatResponse, LiteLLMParamsBody, ThinkingParam from passthrough_client import PassthroughClient -pytestmark = pytest.mark.e2e +pytestmark = [ + pytest.mark.e2e, + pytest.mark.e2e_coverage( + module="core_llms", + endpoint="/chat/completions", + provider="deepseek", + params=["reasoning"], + ), +] REASONER = "deepseek/deepseek-reasoner" PROMPT = "What is 17 + 26? Answer with just the number." diff --git a/tests/e2e/llm_translation/test_embeddings_endpoint_e2e.py b/tests/e2e/llm_translation/test_embeddings_endpoint_e2e.py index 56f2de8bd4f..e0382d378a0 100644 --- a/tests/e2e/llm_translation/test_embeddings_endpoint_e2e.py +++ b/tests/e2e/llm_translation/test_embeddings_endpoint_e2e.py @@ -16,7 +16,15 @@ from endpoints_client import EmbeddingsResult, EndpointsClient from lifecycle import ResourceManager from models import LiteLLMParamsBody -pytestmark = pytest.mark.e2e +pytestmark = [ + pytest.mark.e2e, + pytest.mark.e2e_coverage( + module="non_core_llms", + endpoint="/v1/embeddings", + provider="openai", + params=["embeddings"], + ), +] class TestEmbeddingsEndpoint: @@ -27,7 +35,8 @@ class TestEmbeddingsEndpoint: model_id = endpoints_client.create_model( model, LiteLLMParamsBody( - model="openai/text-embedding-3-small", api_key="os.environ/OPENAI_API_KEY" + model="openai/text-embedding-3-small", + api_key="os.environ/OPENAI_API_KEY", ), ) resources.defer(lambda: endpoints_client.delete_model(model_id)) @@ -36,7 +45,9 @@ class TestEmbeddingsEndpoint: result = endpoints_client.embeddings(key, model, "Say this is a test!") require_successful_call(result) parsed = EmbeddingsResult.model_validate_json(result.body) - assert parsed.first_vector, f"/embeddings returned no vector: {result.body[:300]}" - assert any(component != 0.0 for component in parsed.first_vector), ( - f"embedding vector is all zeros: {result.body[:300]}" - ) + assert ( + parsed.first_vector + ), f"/embeddings returned no vector: {result.body[:300]}" + assert any( + component != 0.0 for component in parsed.first_vector + ), f"embedding vector is all zeros: {result.body[:300]}" diff --git a/tests/e2e/llm_translation/test_image_generation_e2e.py b/tests/e2e/llm_translation/test_image_generation_e2e.py index 4d2211f3be4..098785f5d87 100644 --- a/tests/e2e/llm_translation/test_image_generation_e2e.py +++ b/tests/e2e/llm_translation/test_image_generation_e2e.py @@ -15,7 +15,15 @@ from endpoints_client import EndpointsClient, ImagesResult from lifecycle import ResourceManager from models import LiteLLMParamsBody -pytestmark = pytest.mark.e2e +pytestmark = [ + pytest.mark.e2e, + pytest.mark.e2e_coverage( + module="non_core_llms", + endpoint="/v1/images/generations", + provider="openai", + params=["image_generation"], + ), +] class TestImageGeneration: @@ -37,6 +45,6 @@ class TestImageGeneration: parsed = ImagesResult.model_validate_json(result.body) assert parsed.data, f"/images/generations returned no data: {result.body[:300]}" first = parsed.data[0] - assert first.b64_json or first.url, ( - f"generated image has neither b64_json nor url: {result.body[:300]}" - ) + assert ( + first.b64_json or first.url + ), f"generated image has neither b64_json nor url: {result.body[:300]}" diff --git a/tests/e2e/llm_translation/test_messages_e2e.py b/tests/e2e/llm_translation/test_messages_e2e.py index b0a48f22118..219988a8bfd 100644 --- a/tests/e2e/llm_translation/test_messages_e2e.py +++ b/tests/e2e/llm_translation/test_messages_e2e.py @@ -15,7 +15,15 @@ from endpoints_client import EndpointsClient, MessagesResult from lifecycle import ResourceManager from models import LiteLLMParamsBody -pytestmark = pytest.mark.e2e +pytestmark = [ + pytest.mark.e2e, + pytest.mark.e2e_coverage( + module="core_llms", + endpoint="/v1/messages", + provider="anthropic", + params=["messages_basic"], + ), +] class TestAnthropicMessages: @@ -26,7 +34,8 @@ class TestAnthropicMessages: model_id = endpoints_client.create_model( model, LiteLLMParamsBody( - model="anthropic/claude-haiku-4-5", api_key="os.environ/ANTHROPIC_API_KEY" + model="anthropic/claude-haiku-4-5", + api_key="os.environ/ANTHROPIC_API_KEY", ), ) resources.defer(lambda: endpoints_client.delete_model(model_id)) @@ -36,4 +45,6 @@ class TestAnthropicMessages: require_successful_call(result) parsed = MessagesResult.model_validate_json(result.body) assert parsed.role == "assistant", f"unexpected role: {result.body[:300]}" - assert parsed.text.strip(), f"/v1/messages returned no text: {result.body[:300]}" + assert ( + parsed.text.strip() + ), f"/v1/messages returned no text: {result.body[:300]}" diff --git a/tests/e2e/llm_translation/test_ocr_rust_e2e.py b/tests/e2e/llm_translation/test_ocr_rust_e2e.py index 361bb5126a7..cad141242ee 100644 --- a/tests/e2e/llm_translation/test_ocr_rust_e2e.py +++ b/tests/e2e/llm_translation/test_ocr_rust_e2e.py @@ -26,7 +26,15 @@ from endpoints_client import EndpointsClient from lifecycle import ResourceManager from models import LiteLLMParamsBody, OcrBody, OcrDocument, OcrResponse -pytestmark = pytest.mark.e2e +pytestmark = [ + pytest.mark.e2e, + pytest.mark.e2e_coverage( + module="non_core_llms", + endpoint="/chat/completions", + provider="multiple", + params=["ocr"], + ), +] # Tiny in-repo fixtures served via jsdelivr (sha-pinned, immutable) so the request # bodies stay stable across runs. @@ -146,12 +154,19 @@ def _assert_ocr_document(response: OcrResponse) -> None: class TestRustOcrGateway: @pytest.mark.parametrize("case", RUST_OCR_CASES, ids=_CASE_IDS) def test_rust_ocr_response( - self, endpoints_client: EndpointsClient, resources: ResourceManager, case: _OcrCase + self, + endpoints_client: EndpointsClient, + resources: ResourceManager, + case: _OcrCase, ) -> None: model = f"rust-ocr-{case.suffix}-{unique_marker()}" model_id = endpoints_client.create_model(model, case.provider.litellm_params()) resources.defer(lambda: endpoints_client.delete_model(model_id)) key = resources.key() - response = unwrap(endpoints_client.gateway.ocr(key, OcrBody(model=model, document=case.document))) + response = unwrap( + endpoints_client.gateway.ocr( + key, OcrBody(model=model, document=case.document) + ) + ) _assert_ocr_document(response) diff --git a/tests/e2e/llm_translation/test_passthrough_e2e.py b/tests/e2e/llm_translation/test_passthrough_e2e.py index 37d55c665b3..36baf3e9ccb 100644 --- a/tests/e2e/llm_translation/test_passthrough_e2e.py +++ b/tests/e2e/llm_translation/test_passthrough_e2e.py @@ -25,10 +25,20 @@ from passthrough_client import ( PassthroughClient, ) -pytestmark = pytest.mark.e2e +pytestmark = [ + pytest.mark.e2e, + pytest.mark.e2e_coverage( + module="core_llms", + endpoint="/v1/messages", + provider="anthropic", + params=["passthrough", "cost_logging", "streaming", "tool_calling"], + ), +] -def _fetch_cost_breakdown(client: PassthroughClient, result: StreamingResponse) -> SpendLogRow: +def _fetch_cost_breakdown( + client: PassthroughClient, result: StreamingResponse +) -> SpendLogRow: """The passthrough call's logged row, polled until it carries a cost. Asserts (not skips) that a 2xx passthrough call produced a costed row - the diff --git a/tests/e2e/llm_translation/test_provider_features_e2e.py b/tests/e2e/llm_translation/test_provider_features_e2e.py index cf05a4306b4..f68d611b008 100644 --- a/tests/e2e/llm_translation/test_provider_features_e2e.py +++ b/tests/e2e/llm_translation/test_provider_features_e2e.py @@ -30,7 +30,15 @@ from lifecycle import ResourceManager from models import ChatBody, ChatMessage, ChatResponse, LiteLLMParamsBody from passthrough_client import PassthroughClient -pytestmark = pytest.mark.e2e +pytestmark = [ + pytest.mark.e2e, + pytest.mark.e2e_coverage( + module="core_llms", + endpoint="/chat/completions", + provider="multiple", + params=["service_tier", "prompt_cache"], + ), +] SERVICE_TIER = "flex" CACHE_MIN_READ_TOKENS = 1 @@ -76,9 +84,6 @@ def post_chat(client: PassthroughClient, key: str, body: BaseModel) -> ChatRespo class TestServiceTier: - @pytest.mark.covers( - "llm.chat_completions.openai.service_tier.nonstream.works", exercised_on=[] - ) def test_openai_service_tier_is_echoed( self, client: PassthroughClient, resources: ResourceManager ) -> None: @@ -110,10 +115,6 @@ class TestServiceTier: class TestPromptCaching: - @pytest.mark.covers( - "llm.chat_completions.bedrock_converse.prompt_cache_5m.nonstream.works", - exercised_on=[], - ) def test_bedrock_cache_control_produces_cache_read( self, client: PassthroughClient, resources: ResourceManager ) -> None: diff --git a/tests/e2e/llm_translation/test_rerank_e2e.py b/tests/e2e/llm_translation/test_rerank_e2e.py index 4b30ac1ea5c..75567475670 100644 --- a/tests/e2e/llm_translation/test_rerank_e2e.py +++ b/tests/e2e/llm_translation/test_rerank_e2e.py @@ -15,7 +15,15 @@ from endpoints_client import EndpointsClient, RerankResult from lifecycle import ResourceManager from models import LiteLLMParamsBody -pytestmark = pytest.mark.e2e +pytestmark = [ + pytest.mark.e2e, + pytest.mark.e2e_coverage( + module="non_core_llms", + endpoint="/rerank", + provider="cohere", + params=["rerank"], + ), +] DOCUMENTS = [ "Carson City is the capital city of the American state of Nevada.", @@ -32,7 +40,9 @@ class TestRerank: model = f"e2e-rerank-{unique_marker()}" model_id = endpoints_client.create_model( model, - LiteLLMParamsBody(model="cohere/rerank-v3.5", api_key="os.environ/COHERE_API_KEY"), + LiteLLMParamsBody( + model="cohere/rerank-v3.5", api_key="os.environ/COHERE_API_KEY" + ), ) resources.defer(lambda: endpoints_client.delete_model(model_id)) key = resources.key() @@ -44,6 +54,6 @@ class TestRerank: parsed = RerankResult.model_validate_json(result.body) assert parsed.results, f"/rerank returned no results: {result.body[:300]}" assert len(parsed.results) <= 3, f"top_n=3 not honored: {result.body[:300]}" - assert parsed.results[0].relevance_score is not None, ( - f"top rerank result has no relevance_score: {result.body[:300]}" - ) + assert ( + parsed.results[0].relevance_score is not None + ), f"top rerank result has no relevance_score: {result.body[:300]}" diff --git a/tests/e2e/llm_translation/test_responses_e2e.py b/tests/e2e/llm_translation/test_responses_e2e.py index 743de79880f..c6f7bdeeffc 100644 --- a/tests/e2e/llm_translation/test_responses_e2e.py +++ b/tests/e2e/llm_translation/test_responses_e2e.py @@ -15,7 +15,15 @@ from endpoints_client import EndpointsClient, ResponsesResult from lifecycle import ResourceManager from models import LiteLLMParamsBody -pytestmark = pytest.mark.e2e +pytestmark = [ + pytest.mark.e2e, + pytest.mark.e2e_coverage( + module="core_llms", + endpoint="/v1/responses", + provider="openai", + params=["responses_basic"], + ), +] class TestResponses: @@ -25,7 +33,9 @@ class TestResponses: model = f"e2e-responses-{unique_marker()}" model_id = endpoints_client.create_model( model, - LiteLLMParamsBody(model="openai/gpt-4o-mini", api_key="os.environ/OPENAI_API_KEY"), + LiteLLMParamsBody( + model="openai/gpt-4o-mini", api_key="os.environ/OPENAI_API_KEY" + ), ) resources.defer(lambda: endpoints_client.delete_model(model_id)) key = resources.key() @@ -33,4 +43,6 @@ class TestResponses: result = endpoints_client.responses(key, model, "reply with one word") require_successful_call(result) parsed = ResponsesResult.model_validate_json(result.body) - assert parsed.text.strip(), f"/responses returned no output text: {result.body[:300]}" + assert ( + parsed.text.strip() + ), f"/responses returned no output text: {result.body[:300]}" diff --git a/tests/e2e/llm_translation/test_vertex_passthrough_e2e.py b/tests/e2e/llm_translation/test_vertex_passthrough_e2e.py index 78d2bb358d2..8b84a7f47c7 100644 --- a/tests/e2e/llm_translation/test_vertex_passthrough_e2e.py +++ b/tests/e2e/llm_translation/test_vertex_passthrough_e2e.py @@ -33,7 +33,15 @@ from lifecycle import ResourceManager from models import SpendLogRow from passthrough_client import PassthroughClient -pytestmark = pytest.mark.e2e +pytestmark = [ + pytest.mark.e2e, + pytest.mark.e2e_coverage( + module="non_core_llms", + endpoint="/vertex_ai/*", + provider="vertex_ai", + params=["passthrough"], + ), +] VERTEX_MODEL = "gemini-2.5-flash" # The added deployment's region and the passthrough URL's region are the same constant, @@ -148,7 +156,9 @@ class TestVertexPassthroughSpendTracking: vertex_credentials: str, ) -> None: model_name = f"e2e-vertex-pt-{unique_marker()}" - model_id = _add_vertex_passthrough_model(client, model_name, vertex_project, vertex_credentials) + model_id = _add_vertex_passthrough_model( + client, model_name, vertex_project, vertex_credentials + ) resources.defer(lambda: _delete_model(client, model_id)) result = client.vertex_generate( @@ -159,8 +169,12 @@ class TestVertexPassthroughSpendTracking: text=f"reply with one word {unique_marker()}", ) require_successful_call(result) - assert '"candidates"' in result.body, f"vertex passthrough returned no candidates: {result.body[:300]}" + assert ( + '"candidates"' in result.body + ), f"vertex passthrough returned no candidates: {result.body[:300]}" row = _costed_row(client, result.call_id) - assert row.custom_llm_provider == "vertex_ai", f"passthrough spend logged under the wrong provider: {row}" + assert ( + row.custom_llm_provider == "vertex_ai" + ), f"passthrough spend logged under the wrong provider: {row}" assert "gemini" in (row.model or ""), f"unexpected model in spend log: {row}" diff --git a/tests/e2e/logging/test_prometheus_cardinality_e2e.py b/tests/e2e/logging/test_prometheus_cardinality_e2e.py index 163293a3009..645398dc83c 100644 --- a/tests/e2e/logging/test_prometheus_cardinality_e2e.py +++ b/tests/e2e/logging/test_prometheus_cardinality_e2e.py @@ -18,13 +18,23 @@ from __future__ import annotations import time import pytest -from prometheus_client.parser import text_string_to_metric_families from e2e_config import unique_marker from lifecycle import ResourceManager from logging_client import LoggingClient -pytestmark = pytest.mark.e2e +pytestmark = [ + pytest.mark.e2e, + pytest.mark.e2e_coverage( + module="logging", + endpoint="logging", + provider="prometheus", + params=["prometheus_cardinality"], + ), +] + +prometheus_parser = pytest.importorskip("prometheus_client.parser") +text_string_to_metric_families = prometheus_parser.text_string_to_metric_families DRIVER_MODEL = "gemini-2.5-flash" REQUESTS_METRIC = "litellm_requests_metric_total" @@ -43,7 +53,6 @@ def _aliases_in_metric(exposition: str, metric: str, label: str) -> frozenset[st class TestPrometheusPerKeyCardinality: - @pytest.mark.covers("logging.prometheus.success.exports_metric", exercised_on=[]) def test_distinct_key_aliases_produce_distinct_series( self, client: LoggingClient, resources: ResourceManager ) -> None: @@ -52,13 +61,17 @@ class TestPrometheusPerKeyCardinality: key = client.key_with_alias(alias, models=[DRIVER_MODEL]) resources.defer(lambda k=key: client.delete_key(k)) response = client.chat(key, DRIVER_MODEL, f"reply with one word {alias}") - assert response.model, f"driver call for {alias} returned no model: {response}" + assert ( + response.model + ), f"driver call for {alias} returned no model: {response}" wanted = frozenset(aliases) deadline = time.monotonic() + client.gateway.poll_timeout seen: frozenset[str] = frozenset() while time.monotonic() < deadline: - seen = _aliases_in_metric(client.scrape_metrics(), REQUESTS_METRIC, ALIAS_LABEL) + seen = _aliases_in_metric( + client.scrape_metrics(), REQUESTS_METRIC, ALIAS_LABEL + ) if wanted <= seen: break time.sleep(client.gateway.poll_interval) diff --git a/tests/e2e/management/test_key_models_dropdown_e2e.py b/tests/e2e/management/test_key_models_dropdown_e2e.py index f0ba21699e0..621d6fa2ec3 100644 --- a/tests/e2e/management/test_key_models_dropdown_e2e.py +++ b/tests/e2e/management/test_key_models_dropdown_e2e.py @@ -21,16 +21,36 @@ from models import KeyGenerateBody, TeamNewBody pytest.importorskip("playwright.sync_api", reason="playwright not installed") -from playwright.sync_api import Locator, Page, expect # noqa: E402 # import must follow the importorskip guard above +from playwright.sync_api import ( + Locator, + Page, + expect, +) # noqa: E402 # import must follow the importorskip guard above + +pytestmark = [ + pytest.mark.e2e, + pytest.mark.e2e_coverage( + module="management", + endpoint="/key/*", + provider="proxy", + params=["models_dropdown"], + ), +] def _form_item(page: Page, label: str) -> Locator: - return page.locator(".ant-form-item").filter(has=page.get_by_text(label, exact=True)).first + return ( + page.locator(".ant-form-item") + .filter(has=page.get_by_text(label, exact=True)) + .first + ) def _open_dropdown(page: Page, label: str) -> Locator: _form_item(page, label).locator(".ant-select-selector").first.click() - dropdown = page.locator(".ant-select-dropdown:not(.ant-select-dropdown-hidden)").last + dropdown = page.locator( + ".ant-select-dropdown:not(.ant-select-dropdown-hidden)" + ).last expect(dropdown).to_be_visible() return dropdown @@ -38,7 +58,9 @@ def _open_dropdown(page: Page, label: str) -> Locator: def _models_dropdown_texts(page: Page, must_contain: str) -> list[str]: dropdown = _open_dropdown(page, "Models") expect( - dropdown.locator(".ant-select-item-option-content", has_text=must_contain).first, + dropdown.locator( + ".ant-select-item-option-content", has_text=must_contain + ).first, f"{must_contain!r} never appeared in the Models dropdown; the proxy must serve it " f"(see the model_list in tests/e2e/docker-compose.yml)", ).to_be_visible() @@ -56,15 +78,23 @@ def _select_team(page: Page, alias: str) -> None: def _submit_create_modal(page: Page, sentinel_label: str) -> str: - dropdown = page.locator(".ant-select-dropdown:not(.ant-select-dropdown-hidden)").last - dropdown.locator(".ant-select-item-option-content", has_text=sentinel_label).first.click() + dropdown = page.locator( + ".ant-select-dropdown:not(.ant-select-dropdown-hidden)" + ).last + dropdown.locator( + ".ant-select-item-option-content", has_text=sentinel_label + ).first.click() page.keyboard.press("Escape") - _form_item(page, "Key Name").locator("input").first.fill(f"e2e-ui-key-{unique_marker()}") + _form_item(page, "Key Name").locator("input").first.fill( + f"e2e-ui-key-{unique_marker()}" + ) page.get_by_role("button", name="Create Key", exact=True).click() expect(page.get_by_text("Save your Key")).to_be_visible() key = page.locator(".ant-modal pre").last.inner_text().strip() - assert key.startswith("sk-"), f"expected the created key in the success modal, got {key!r}" + assert key.startswith( + "sk-" + ), f"expected the created key in the success modal, got {key!r}" return key @@ -76,31 +106,43 @@ def _open_key_edit_form(page: Page, key_alias: str) -> None: expect(_form_item(page, "Models")).to_be_visible() -def _provision_team(client: ManagementClient, resources: ResourceManager, alias: str) -> str: - team_id = client.create_team(TeamNewBody(team_alias=alias, models=["all-proxy-models", "gpt-5.5"])) +def _provision_team( + client: ManagementClient, resources: ResourceManager, alias: str +) -> str: + team_id = client.create_team( + TeamNewBody(team_alias=alias, models=["all-proxy-models", "gpt-5.5"]) + ) resources.defer(lambda: client.delete_team(team_id)) return team_id def _provision_key( - client: ManagementClient, resources: ResourceManager, alias: str, team_id: str | None = None + client: ManagementClient, + resources: ResourceManager, + alias: str, + team_id: str | None = None, ) -> str: - key = client.gateway.generate_key(KeyGenerateBody(key_alias=alias, models=["gpt-5.5"], team_id=team_id)) + key = client.gateway.generate_key( + KeyGenerateBody(key_alias=alias, models=["gpt-5.5"], team_id=team_id) + ) resources.defer(lambda: client.gateway.delete_key(key)) return key @pytest.mark.e2e class TestKeyModelsDropdownUI: - @pytest.mark.covers("mgmt.key.generate.happy_path", exercised_on=[]) def test_create_teamless_key_offers_proxy_scope_and_persists( self, ui_page: Page, client: ManagementClient, resources: ResourceManager ) -> None: _open_create_key_modal(ui_page) options = _models_dropdown_texts(ui_page, must_contain="gpt-5.5") - assert "All Proxy Models" in options, f"teamless create lost 'All Proxy Models': {options}" - assert "All Team Models" not in options, f"teamless create offered 'All Team Models': {options}" + assert ( + "All Proxy Models" in options + ), f"teamless create lost 'All Proxy Models': {options}" + assert ( + "All Team Models" not in options + ), f"teamless create offered 'All Team Models': {options}" key = _submit_create_modal(ui_page, sentinel_label="All Proxy Models") resources.defer(lambda: client.gateway.delete_key(key)) @@ -109,7 +151,6 @@ class TestKeyModelsDropdownUI: assert info.models == ["all-proxy-models"], f"persisted models {info.models}" assert info.team_id is None, f"teamless key persisted with team {info.team_id}" - @pytest.mark.covers("mgmt.key.generate.happy_path", exercised_on=[]) def test_create_team_key_offers_team_scope_and_persists( self, ui_page: Page, client: ManagementClient, resources: ResourceManager ) -> None: @@ -120,18 +161,25 @@ class TestKeyModelsDropdownUI: _select_team(ui_page, team_alias) options = _models_dropdown_texts(ui_page, must_contain="All Team Models") - assert "gpt-5.5" in options, f"team key create lost the team's own model: {options}" - assert "All Proxy Models" not in options, f"team key create offered 'All Proxy Models': {options}" - assert "all-proxy-models" not in options, f"team key create offered the raw sentinel: {options}" + assert ( + "gpt-5.5" in options + ), f"team key create lost the team's own model: {options}" + assert ( + "All Proxy Models" not in options + ), f"team key create offered 'All Proxy Models': {options}" + assert ( + "all-proxy-models" not in options + ), f"team key create offered the raw sentinel: {options}" key = _submit_create_modal(ui_page, sentinel_label="All Team Models") resources.defer(lambda: client.gateway.delete_key(key)) info = client.gateway.key_info(key) assert info.models == ["all-team-models"], f"persisted models {info.models}" - assert info.team_id == team_id, f"persisted team {info.team_id}, expected {team_id}" + assert ( + info.team_id == team_id + ), f"persisted team {info.team_id}, expected {team_id}" - @pytest.mark.covers("mgmt.key.update.happy_path", exercised_on=[]) def test_edit_teamless_key_offers_proxy_scope( self, ui_page: Page, client: ManagementClient, resources: ResourceManager ) -> None: @@ -141,10 +189,13 @@ class TestKeyModelsDropdownUI: _open_key_edit_form(ui_page, key_alias) options = _models_dropdown_texts(ui_page, must_contain="gpt-5.5") - assert "All Proxy Models" in options, f"teamless edit lost 'All Proxy Models': {options}" - assert "All Team Models" not in options, f"teamless edit offered 'All Team Models': {options}" + assert ( + "All Proxy Models" in options + ), f"teamless edit lost 'All Proxy Models': {options}" + assert ( + "All Team Models" not in options + ), f"teamless edit offered 'All Team Models': {options}" - @pytest.mark.covers("mgmt.key.update.happy_path", exercised_on=[]) def test_edit_team_key_offers_team_scope_only( self, ui_page: Page, client: ManagementClient, resources: ResourceManager ) -> None: @@ -156,6 +207,12 @@ class TestKeyModelsDropdownUI: _open_key_edit_form(ui_page, key_alias) options = _models_dropdown_texts(ui_page, must_contain="All Team Models") - assert "gpt-5.5" in options, f"team key edit lost the team's own model: {options}" - assert "All Proxy Models" not in options, f"team key edit offered 'All Proxy Models': {options}" - assert "all-proxy-models" not in options, f"team key edit offered the raw sentinel: {options}" + assert ( + "gpt-5.5" in options + ), f"team key edit lost the team's own model: {options}" + assert ( + "All Proxy Models" not in options + ), f"team key edit offered 'All Proxy Models': {options}" + assert ( + "all-proxy-models" not in options + ), f"team key edit offered the raw sentinel: {options}" diff --git a/tests/e2e/management/test_management_e2e.py b/tests/e2e/management/test_management_e2e.py index cbd5db0d59f..1a40f33441c 100644 --- a/tests/e2e/management/test_management_e2e.py +++ b/tests/e2e/management/test_management_e2e.py @@ -24,7 +24,15 @@ from management_client import ( ) from models import KeyGenerateBody, OrgNewBody, TeamNewBody, UserNewBody -pytestmark = pytest.mark.e2e +pytestmark = [ + pytest.mark.e2e, + pytest.mark.e2e_coverage( + module='management', + endpoint='/key/*', + provider='proxy', + params=['key_crud', 'team_crud', 'user_crud', 'organization_crud', 'route_permission'], + ), +] def _poll[T](client: ManagementClient, attempt: Callable[[], T | None], failure: str) -> T: deadline = time.monotonic() + client.gateway.poll_timeout @@ -103,7 +111,6 @@ def _poll_model_access_granted(client: ManagementClient, key: str, model: str) - class TestKeyRoutes: - @pytest.mark.covers("mgmt.key.generate.persists") def test_generate_persists_to_key_info_and_scopes_chat( self, client: ManagementClient, resources: ResourceManager ) -> None: @@ -128,7 +135,6 @@ class TestKeyRoutes: client.chat_status(key, "gpt-5.5", f"say hi {unique_marker()}"), "gpt-5.5" ) - @pytest.mark.covers("mgmt.key.update.persists") def test_update_models_persists_and_flips_enforcement( self, client: ManagementClient, resources: ResourceManager ) -> None: @@ -148,7 +154,6 @@ class TestKeyRoutes: _poll_model_access_granted(client, key, "gpt-5.5") _poll_chat_denied(client, key, "gemini-2.5-flash") - @pytest.mark.covers("mgmt.key.delete.persists") def test_delete_revokes_the_key_on_chat(self, client: ManagementClient, resources: ResourceManager) -> None: """The teardown's deferred delete fires again on the already-deleted key by design: the deferred cleanup must survive this test failing before the @@ -167,7 +172,6 @@ class TestKeyRoutes: class TestTeamRoutes: - @pytest.mark.covers("mgmt.team.new.persists") def test_new_persists_to_team_info_and_binds_keys( self, client: ManagementClient, resources: ResourceManager ) -> None: @@ -186,7 +190,6 @@ class TestTeamRoutes: f"key generated under team {team_id} carries team_id {key_info.team_id!r} in /key/info" ) - @pytest.mark.covers("mgmt.team.member_add.persists") def test_member_add_and_delete_persist_to_team_info( self, client: ManagementClient, resources: ResourceManager ) -> None: @@ -212,7 +215,6 @@ class TestTeamRoutes: class TestUserRoutes: - @pytest.mark.covers("mgmt.user.new.happy_path") def test_new_persists_to_user_info(self, client: ManagementClient, resources: ResourceManager) -> None: email = f"e2e-mgmt-{unique_marker()}@example.com" user_id = _create_user(client, resources, UserNewBody(user_email=email, user_role="internal_user")) @@ -225,7 +227,6 @@ class TestUserRoutes: class TestOrganizationRoutes: - @pytest.mark.covers("mgmt.organization.new.happy_path") def test_new_persists_to_organization_info( self, client: ManagementClient, resources: ResourceManager ) -> None: @@ -252,7 +253,6 @@ def _assert_route_forbidden(route: str, outcome: StreamingResponse) -> None: class TestManagementRoutePermissions: - @pytest.mark.covers("other.auth.virtual_key.route_permission_enforced") def test_llm_only_key_forbidden_from_management_writes( self, client: ManagementClient, resources: ResourceManager ) -> None: diff --git a/tests/e2e/pytest.ini b/tests/e2e/pytest.ini index 7799f6b16a2..055ebe56761 100644 --- a/tests/e2e/pytest.ini +++ b/tests/e2e/pytest.ini @@ -5,3 +5,4 @@ addopts = --strict-markers --strict-config markers = e2e: live test that requires a running proxy and real provider keys + e2e_coverage(module, endpoint, provider, params): structured e2e coverage metadata diff --git a/tests/e2e/rate_limits/README.md b/tests/e2e/rate_limits/README.md new file mode 100644 index 00000000000..f4ed57a0e65 --- /dev/null +++ b/tests/e2e/rate_limits/README.md @@ -0,0 +1,13 @@ +# Rate Limits E2E Suite + +Put rate-limit enforcement tests here when the primary behavior under test is a +limit being applied, reset, or bypassed across keys, teams, models, tags, or +endpoint families. + +If the primary behavior is provider translation or an LLM endpoint contract, keep +the test under `tests/e2e/llm_translation/` instead. + +Coverage for this module is declared directly on tests with +`@pytest.mark.e2e_coverage(...)`. Use `module="rate_limits"`, the exercised +endpoint, `provider="proxy"`, and params such as `key_rpm_limit`, +`team_tpm_limit`, or `reset_window`. diff --git a/tests/e2e/spend_tracking/test_spend_routes.py b/tests/e2e/spend_tracking/test_spend_routes.py index 9b4eaefae34..54be3e36439 100644 --- a/tests/e2e/spend_tracking/test_spend_routes.py +++ b/tests/e2e/spend_tracking/test_spend_routes.py @@ -23,7 +23,15 @@ import pytest from models import DateRangeParams from spend_e2e_client import SpendClient -pytestmark = pytest.mark.e2e +pytestmark = [ + pytest.mark.e2e, + pytest.mark.e2e_coverage( + module="spend_tracking", + endpoint="/spend/*", + provider="proxy", + params=["spend_routes"], + ), +] # Verified present and responsive on a live proxy. One per row of the spend # surface: key / user / team / org / customer aggregation, model-cost, tags, diff --git a/tests/e2e/spend_tracking/test_spend_tracking_e2e.py b/tests/e2e/spend_tracking/test_spend_tracking_e2e.py index 3495011eab6..7ff1ca97161 100644 --- a/tests/e2e/spend_tracking/test_spend_tracking_e2e.py +++ b/tests/e2e/spend_tracking/test_spend_tracking_e2e.py @@ -26,7 +26,15 @@ from lifecycle import ResourceManager from models import ChatResponse, SpendLogs, SpendLogsParams from spend_e2e_client import SpendClient, SpendLogRow, is_ok, unique_marker, unwrap -pytestmark = pytest.mark.e2e +pytestmark = [ + pytest.mark.e2e, + pytest.mark.e2e_coverage( + module="spend_tracking", + endpoint="/spend/*", + provider="proxy", + params=["spend_logging", "spend_filters"], + ), +] def _approx_equal(actual: float, expected: float) -> bool: @@ -237,9 +245,9 @@ def test_burst_of_concurrent_calls_loses_no_spend( f"rows lost under concurrency: {_summarize(rows)}" ) request_ids = [r.request_id for r in costed] - assert len(set(request_ids)) == len(request_ids), ( - f"concurrent rows collapsed onto shared request_ids: {_summarize(rows)}" - ) + assert len(set(request_ids)) == len( + request_ids + ), f"concurrent rows collapsed onto shared request_ids: {_summarize(rows)}" logs_total = sum((r.spend or 0) for r in rows) key_spend = client.poll_key_spend(scoped_key, minimum=logs_total * 0.999) @@ -288,9 +296,9 @@ def test_spend_logs_v2_pagination_caps_pages_and_keeps_total( api_key=hashed_key, page=first.total_pages + 7, page_size=1 ) assert beyond.data == [], f"out-of-range page returned rows: {beyond.data}" - assert beyond.total == first.total, ( - f"out-of-range page changed the total: {beyond.total} != {first.total}" - ) + assert ( + beyond.total == first.total + ), f"out-of-range page changed the total: {beyond.total} != {first.total}" nomatch = client.spend_logs_page( api_key=f"sk-no-such-key-{unique_marker()}", page=1, page_size=1 @@ -346,12 +354,12 @@ def test_tag_spend_matches_sum_of_tagged_logs( entry = client.poll_tag_spend(tag, minimum=logs_total * 0.999) assert entry is not None, f"tag {tag!r} never appeared in /spend/tags" - assert _approx_equal(entry.total_spend or 0, logs_total), ( - f"/spend/tags total_spend {entry} != sum of tagged rows {logs_total}" - ) - assert (entry.log_count or 0) == len(tagged), ( - f"/spend/tags log_count {entry.log_count} != tagged rows {len(tagged)}" - ) + assert _approx_equal( + entry.total_spend or 0, logs_total + ), f"/spend/tags total_spend {entry} != sum of tagged rows {logs_total}" + assert (entry.log_count or 0) == len( + tagged + ), f"/spend/tags log_count {entry.log_count} != tagged rows {len(tagged)}" def test_end_user_spend_attributed_on_row( @@ -396,7 +404,9 @@ def test_each_model_on_a_shared_key_gets_its_own_row( "claude-haiku-4-5" in m for m in costed ) - rows = client.poll_logs_for_key(scoped_key, min_rows=2, predicate=both_models_costed) + rows = client.poll_logs_for_key( + scoped_key, min_rows=2, predicate=both_models_costed + ) gemini_row = _require_row( rows, lambda r: "gemini-2.5-flash" in (r.model or ""), "for the gemini call" ) @@ -404,8 +414,12 @@ def test_each_model_on_a_shared_key_gets_its_own_row( rows, lambda r: "claude-haiku-4-5" in (r.model or ""), "for the claude call" ) - assert (gemini_row.spend or 0) > 0, f"gemini row should cost > 0: {_summarize(rows)}" - assert (claude_row.spend or 0) > 0, f"claude row should cost > 0: {_summarize(rows)}" + assert ( + gemini_row.spend or 0 + ) > 0, f"gemini row should cost > 0: {_summarize(rows)}" + assert ( + claude_row.spend or 0 + ) > 0, f"claude row should cost > 0: {_summarize(rows)}" assert ( gemini_row.request_id != claude_row.request_id ), f"two distinct calls collapsed onto one request_id: {_summarize(rows)}" @@ -458,7 +472,10 @@ def test_spend_logs_endpoint_returns_spend( call's nonzero spend must surface before the deadline.""" unwrap( client.chat( - scoped_key, "gemini-2.5-flash", f"spend logs {unique_marker()}", max_tokens=16 + scoped_key, + "gemini-2.5-flash", + f"spend logs {unique_marker()}", + max_tokens=16, ) ) @@ -471,7 +488,9 @@ def test_spend_logs_endpoint_returns_spend( params=SpendLogsParams(api_key=scoped_key), response_type=SpendLogs, ) - assert isinstance(result, Success), f"/spend/logs did not return 200 OK: {result}" + assert isinstance( + result, Success + ), f"/spend/logs did not return 200 OK: {result}" rows = result.data.root if sum((r.spend or 0) for r in rows) > 0: return diff --git a/tests/e2e/test_e2e_gateway.py b/tests/e2e/test_e2e_gateway.py index a6dcc6112d6..665f75a1f6c 100644 --- a/tests/e2e/test_e2e_gateway.py +++ b/tests/e2e/test_e2e_gateway.py @@ -1,7 +1,7 @@ """Unit coverage for the Gateway model-management surface (create_model / delete_model). -The batches conftest and several llm_translation tests register deployments at +The llm_translation batches conftest and several llm_translation tests register deployments at runtime through gateway.create_model; when that method went missing, every batch test errored at fixture setup (AttributeError) before a single request reached the proxy. This pins the surface with a typed fake Transport so a rename or @@ -10,9 +10,9 @@ signature drift fails here instead of in a live stage run. from dataclasses import dataclass, field +import pytest from pydantic import BaseModel -from batches.batch_client import BatchClient from e2e_gateway import Gateway from e2e_http import ( AuthHeaders, @@ -22,6 +22,7 @@ from e2e_http import ( StreamingResponse, Success, ) +from llm_translation.batches.batch_client import BatchClient from models import ( LiteLLMParamsBody, ModelDeleteBody, @@ -29,6 +30,13 @@ from models import ( ModelNewResponse, ) +pytestmark = pytest.mark.e2e_coverage( + module="other", + endpoint="e2e_harness", + provider="proxy", + params=["gateway_model_management"], +) + @dataclass class _RecordingTransport: diff --git a/tests/e2e/test_lifecycle.py b/tests/e2e/test_lifecycle.py index d3c559dd2ed..ae8dd7b4306 100644 --- a/tests/e2e/test_lifecycle.py +++ b/tests/e2e/test_lifecycle.py @@ -12,6 +12,13 @@ import pytest from lifecycle import run_case +pytestmark = pytest.mark.e2e_coverage( + module="reliability", + endpoint="e2e_harness", + provider="proxy", + params=["resource_cleanup"], +) + @dataclass class _PartialInitCase: diff --git a/tests/e2e/test_transport.py b/tests/e2e/test_transport.py index c7ce61b90c1..1039b3d954a 100644 --- a/tests/e2e/test_transport.py +++ b/tests/e2e/test_transport.py @@ -3,7 +3,7 @@ Model-management calls (/model/new, /model/delete, /model/info) must go to the control plane: the data-plane gateway does not serve management routes, so a misrouted /model/new 404s and takes down every suite that registers deployments -at runtime (llm_translation, batches, access_control). /models must stay on the +at runtime (llm_translation, access_control). /models must stay on the data plane; it is the OpenAI-compatible list-models route, not a management route. """ @@ -12,6 +12,13 @@ import pytest from transport import is_control_plane_path +pytestmark = pytest.mark.e2e_coverage( + module="other", + endpoint="e2e_harness", + provider="proxy", + params=["split_transport_routing"], +) + @pytest.mark.parametrize( "path", @@ -30,9 +37,9 @@ from transport import is_control_plane_path ], ) def test_management_routes_go_to_the_control_plane(path: str) -> None: - assert is_control_plane_path(path), ( - f"{path} is a management route; sending it to the data plane 404s" - ) + assert is_control_plane_path( + path + ), f"{path} is a management route; sending it to the data plane 404s" @pytest.mark.parametrize( @@ -47,6 +54,6 @@ def test_management_routes_go_to_the_control_plane(path: str) -> None: ], ) def test_llm_routes_stay_on_the_data_plane(path: str) -> None: - assert not is_control_plane_path(path), ( - f"{path} is an LLM route; it must go to the data plane" - ) + assert not is_control_plane_path( + path + ), f"{path} is an LLM route; it must go to the data plane"