From ae05f7d2c1dcb47cbc3d06a633296487f30b477d Mon Sep 17 00:00:00 2001 From: ryan-crabbe-berri Date: Wed, 30 Sep 2026 21:03:21 -0700 Subject: [PATCH] test(e2e): typed per-test metadata for the e2e suite (#42044) * feat(e2e): give e2e tests typed metadata for what they drive @meta(Subject(domain, route, providers, models, capabilities, mode)) declares what a test is about with closed enums, and each field lands in the JUnit report as a property. The quota_management suites are the first to declare it. * docs(e2e): say e2e_metadata avoids litellm, not that it is stdlib-only It already imports pydantic and pytest, both of which the suite needs to collect. The rule that matters is no litellm import * test(e2e): declare models through the constant each test drives 43 @meta declarations in quota_management typed the model name out again, so changing the call would leave the coverage report naming the old model. Each file now has one constant used by both, and a guard fails on any model written as a string literal in @meta * refactor(e2e): set route only when the endpoint is what the test checks A budget or rate-limit test whose chat call only triggers the block now leaves route unset, since its steps already name the call. Tests of an endpoint keep it: budget CRUD, key creation, spend reporting reads, and the per-endpoint spend tests for chat, messages, embeddings, batches and health. The two /spend/logs tests tagged chat_completions are now spend_reporting * refactor(e2e): build the declared properties without mutating a list subject_properties seeded a list and grew it with append and extend. It now flattens one tuple per field, and the plural-name table is a read-only mapping * fix(e2e): tag each spend-route probe with the endpoint it checks The breadth test gave all 33 probes spend_reporting, so /key/list, /user/list, /team/list, /organization/list and /customer/list counted as spend reporting. Each case now carries its own route, with organization and customer management added to Route --- .../test_e2e_junit_report.py | 57 ++++- .../code_coverage_tests/test_e2e_metadata.py | 207 ++++++++++++++++- tests/e2e/AGENTS.md | 23 ++ tests/e2e/conftest.py | 5 + tests/e2e/e2e_metadata.py | 219 +++++++++++++++++- tests/e2e/junit_properties.py | 14 +- tests/e2e/pytest.ini | 1 + .../budgets/test_budget_crud_e2e.py | 19 ++ .../budgets/test_budget_enforcement_e2e.py | 78 ++++++- .../budgets/test_budget_fallback_e2e.py | 10 + .../budgets/test_budget_reset_advances_e2e.py | 50 +++- .../budgets/test_budget_reset_e2e.py | 60 ++++- .../test_model_access_group_budget_e2e.py | 33 +++ .../budgets/test_model_max_budget_e2e.py | 17 ++ .../budgets/test_multi_window_budget_e2e.py | 17 ++ .../budgets/test_soft_budget_e2e.py | 13 +- .../budgets/test_spend_counter_reseed_e2e.py | 9 + .../budgets/test_tag_budget_e2e.py | 12 +- .../budgets/test_team_member_budget_e2e.py | 17 ++ .../test_team_member_budget_isolation_e2e.py | 9 + .../test_team_member_budget_reset_e2e.py | 12 +- .../test_team_multi_window_budget_e2e.py | 24 +- .../test_user_budget_across_keys_e2e.py | 9 + .../test_dynamic_rate_limit_priority_e2e.py | 17 ++ .../ratelimit/test_rate_limit_e2e.py | 33 +++ .../test_redis_backed_ratelimit_e2e.py | 9 + .../test_redis_circuit_breaker_e2e.py | 9 + .../test_tpm_excludes_cached_tokens_e2e.py | 10 + .../spend_tracking/spend_reconciliation.py | 3 +- .../test_cache_cost_accounting_e2e.py | 38 +++ .../spend_tracking/test_cost_headers_e2e.py | 9 + .../test_key_attribution_e2e.py | 43 ++++ .../test_provider_edge_spend_e2e.py | 10 + .../test_service_tier_pricing_e2e.py | 10 + .../spend_tracking/test_spend_routes.py | 29 ++- .../spend_tracking/test_spend_tracking_e2e.py | 182 +++++++++++++-- .../test_team_daily_activity_e2e.py | 23 +- 37 files changed, 1281 insertions(+), 59 deletions(-) diff --git a/tests/code_coverage_tests/test_e2e_junit_report.py b/tests/code_coverage_tests/test_e2e_junit_report.py index f98cc25a2d1..140b9ba6dca 100644 --- a/tests/code_coverage_tests/test_e2e_junit_report.py +++ b/tests/code_coverage_tests/test_e2e_junit_report.py @@ -38,7 +38,7 @@ from collections.abc import Iterator from pathlib import Path import pytest -from e2e_metadata import step +from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta, step FIRST_ATTEMPT_MADE = Path(__file__).with_name("first-attempt-made") @@ -100,6 +100,20 @@ def test_passes_on_the_rerun(key: None) -> None: FIRST_ATTEMPT_MADE.touch() chat(ok=not first_attempt) poll_spend_logs() + + +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.BEDROCK, Provider.ANTHROPIC), + models=("claude-sonnet-4-5", "claude-opus-4-7", "claude-haiku-4-5"), + capabilities=(Capability.VISION, Capability.FUNCTION_CALLING), + mode=Mode.STREAM, + ) +) +def test_declares_two_providers_and_three_models() -> None: + assert Provider.BEDROCK.value == "bedrock" """ WIDE_FINALIZER_SUITE: Final = """ @@ -183,6 +197,15 @@ def pytest_runtest_logreport(report: pytest.TestReport) -> None: out.write(json.dumps([report.nodeid.split("::")[-1], steps]) + "\\n") """ +BARE_STR_SUITE: Final = """ +from e2e_metadata import Subject, meta + + +@meta(Subject(models=("gpt-5.5"))) +def test_never_collected() -> None: + assert Subject is not None +""" + Properties = tuple[tuple[str, str], ...] FailedReport: Final = TypeAdapter(tuple[str, tuple[str, ...]]) @@ -278,7 +301,7 @@ def report(request: pytest.FixtureRequest, tmp_path_factory: pytest.TempPathFact assert xml.exists(), f"the child run wrote no JUnit report:\n{child.stdout}\n{child.stderr}" testsuite: Final = next(ElementTree.parse(xml).getroot().iter("testsuite")) outcomes: Final = {name: testsuite.get(name) for name in ("tests", "failures", "errors", "skipped")} - assert outcomes == {"tests": "6", "failures": "1", "errors": "2", "skipped": "0"}, child.stdout + assert outcomes == {"tests": "7", "failures": "1", "errors": "2", "skipped": "0"}, child.stdout return properties_by_test(testsuite) @@ -340,3 +363,33 @@ def test_a_failed_phase_s_own_report_carries_the_steps(tmp_path: Path) -> None: "test_oauth_dies_on_consent": ("open the consent page",), "test_plain_dies_on_consent": ("open the consent page",), }, child.stdout + + +class TestDeclaredPropertiesReachTheReport: + def test_repeated_provider_model_and_capability_round_trip(self, report: Mapping[str, Properties]) -> None: + declared: Final = tuple( + (prop, value) + for prop, value in report["test_declares_two_providers_and_three_models"] + if prop not in {"package", "covers", "source"} + ) + assert declared == ( + ("domain", "llm-translation"), + ("route", "messages"), + ("provider", "anthropic"), + ("provider", "bedrock"), + ("model", "claude-haiku-4-5"), + ("model", "claude-opus-4-7"), + ("model", "claude-sonnet-4-5"), + ("capability", "function_calling"), + ("capability", "vision"), + ("mode", "stream"), + ) + + +class TestBareStrIsACollectionError: + def test_a_str_where_a_tuple_belongs_fails_collection_and_names_the_fix(self, tmp_path: Path) -> None: + write_suite(tmp_path, {"test_bare_str.py": BARE_STR_SUITE}) + child: Final = run_child_pytest(tmp_path) + assert child.returncode == pytest.ExitCode.INTERRUPTED, child.stdout + assert "Subject.models must be a tuple, got str: 'gpt-5.5'" in child.stdout + assert "models=(x,), not models=(x)" in child.stdout diff --git a/tests/code_coverage_tests/test_e2e_metadata.py b/tests/code_coverage_tests/test_e2e_metadata.py index a18e8300f7c..b1612d3d259 100644 --- a/tests/code_coverage_tests/test_e2e_metadata.py +++ b/tests/code_coverage_tests/test_e2e_metadata.py @@ -1,4 +1,4 @@ -"""The e2e step recorder's edge cases: label templates, dedupe, the cap, nesting, context managers. +"""The e2e test metadata: `@meta(Subject(...))` properties and the step recorder's edge cases. Harness logic, so it lives here rather than under tests/e2e, which holds only tests that drive a live proxy. The harness modules are imported off @@ -18,12 +18,30 @@ import threading import warnings from collections.abc import Callable, Generator, Iterator, Mapping from contextlib import contextmanager +from dataclasses import fields, replace from pathlib import Path from types import UnionType from typing import Final, cast, get_args, get_type_hints import pytest -from e2e_metadata import MASK, MAX_STEPS, STEP_FRAMES, STEPS, StepRecorder, environment_secrets, step +from e2e_metadata import ( + MASK, + MAX_STEPS, + STEP_FRAMES, + STEPS, + Capability, + Domain, + Mode, + Provider, + Route, + StepRecorder, + Subject, + environment_secrets, + meta, + step, + subject_properties, +) +from junit_properties import package_from_nodeid, result_properties, source_from_item from proxy_client import ProxyClient from pydantic import BaseModel, Field from pydantic.fields import FieldInfo @@ -38,6 +56,191 @@ def empty_step_log() -> Generator[None]: STEPS.reset() +def collected_item(request: pytest.FixtureRequest, name: str) -> pytest.Item: + return next(item for item in request.session.items if item.path == request.path and item.name == name) + + +def fixed_prefix(item: pytest.Item, covers: str) -> tuple[tuple[str, str], ...]: + """Spelled out rather than taken from `result_properties`, so a change to either fails a test.""" + return ( + ("package", package_from_nodeid(item.nodeid)), + ("covers", covers), + ("source", source_from_item(item)), + ) + + +class TestSubjectProperties: + """Markers go on via `request.applymarker` so the coverage registry's collect-only pass never sees them.""" + + def test_every_declared_field_becomes_a_property_in_field_order(self, request: pytest.FixtureRequest) -> None: + test = type(self).test_every_declared_field_becomes_a_property_in_field_order + request.applymarker( + meta( + Subject( + domain=Domain.SPEND_BUDGETS, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.GEMINI, Provider.ANTHROPIC), + models=("gemini-2.5-flash", "claude-haiku-4-5"), + capabilities=(Capability.VISION, Capability.FUNCTION_CALLING, Capability.VISION), + mode=Mode.NONSTREAM, + ) + ) + ) + assert subject_properties(collected_item(request, test.__name__)) == ( + ("domain", "spend-budgets"), + ("route", "chat_completions"), + ("provider", "anthropic"), + ("provider", "gemini"), + ("model", "claude-haiku-4-5"), + ("model", "gemini-2.5-flash"), + ("capability", "function_calling"), + ("capability", "vision"), + ("mode", "nonstream"), + ) + + def test_one_provider_with_three_models_pairs_nothing(self, request: pytest.FixtureRequest) -> None: + test = type(self).test_one_provider_with_three_models_pairs_nothing + request.applymarker( + meta( + Subject( + providers=(Provider.BEDROCK,), + models=("claude-sonnet-4-5", "claude-opus-4-7", "claude-haiku-4-5"), + ) + ) + ) + assert subject_properties(collected_item(request, test.__name__)) == ( + ("provider", "bedrock"), + ("model", "claude-haiku-4-5"), + ("model", "claude-opus-4-7"), + ("model", "claude-sonnet-4-5"), + ) + + def test_an_empty_plural_field_emits_nothing(self, request: pytest.FixtureRequest) -> None: + test = type(self).test_an_empty_plural_field_emits_nothing + request.applymarker(meta(Subject(domain=Domain.MANAGEMENT))) + assert subject_properties(collected_item(request, test.__name__)) == (("domain", "management"),) + + def test_scalar_property_names_are_the_dataclass_field_names(self, request: pytest.FixtureRequest) -> None: + test = type(self).test_scalar_property_names_are_the_dataclass_field_names + request.applymarker(meta(Subject(domain=Domain.UNKNOWN, route=Route.HEALTH, mode=Mode.STREAM))) + declared = tuple(field.name for field in fields(Subject)) + emitted = tuple(name for name, _ in subject_properties(collected_item(request, test.__name__))) + assert emitted == tuple(name for name in declared if name in {"domain", "route", "mode"}) + + def test_every_plural_field_is_deduped_and_sorted_at_declaration(self) -> None: + subject = Subject( + providers=(Provider.OPENAI, Provider.ANTHROPIC, Provider.OPENAI), + models=("gpt-5.5", "claude-haiku-4-5", "gpt-5.5"), + capabilities=(Capability.VISION, Capability.REASONING, Capability.VISION), + ) + assert subject.providers == (Provider.ANTHROPIC, Provider.OPENAI) + assert subject.models == ("claude-haiku-4-5", "gpt-5.5") + assert subject.capabilities == (Capability.REASONING, Capability.VISION) + + @pytest.mark.parametrize( + ("field", "value"), + [ + ("models", "gpt-5.5"), + ("models", ["gpt-5.5"]), + ("providers", Provider.OPENAI), + ("providers", [Provider.OPENAI]), + ("capabilities", Capability.VISION), + ("capabilities", frozenset({Capability.VISION})), + ], + ) + def test_a_plural_field_refuses_anything_but_a_tuple(self, field: str, value: object) -> None: + """`replace` is the untyped way in, since the typed constructor would not let the test spell the mistake.""" + with pytest.raises(TypeError, match=rf"Subject\.{field} must be a tuple"): + _ = replace(Subject(), **{field: value}) + + @pytest.mark.parametrize( + ("field", "value", "member_type"), + [ + ("providers", ("openai",), "Provider"), + ("capabilities", ("vision",), "Capability"), + ("models", (5,), "str"), + ], + ) + def test_a_plural_field_refuses_a_member_of_the_wrong_type( + self, field: str, value: object, member_type: str + ) -> None: + with pytest.raises(TypeError, match=rf"Subject\.{field} takes {member_type} members"): + _ = replace(Subject(), **{field: value}) + + def test_a_blank_model_is_dropped_rather_than_refused(self) -> None: + """A blank env override must cost one missing property, not collection of the whole module.""" + assert Subject(models=("", "gpt-5.5")).models == ("gpt-5.5",) + + def test_the_typed_marker_only_ever_appends_to_the_fixed_prefix(self, request: pytest.FixtureRequest) -> None: + test = type(self).test_the_typed_marker_only_ever_appends_to_the_fixed_prefix + request.applymarker(pytest.mark.covers("quota_management.budget.key.blocks_over_limit")) + request.applymarker(meta(Subject(route=Route.SPEND_REPORTING))) + item = collected_item(request, test.__name__) + assert result_properties(item) == fixed_prefix(item, "quota_management.budget.key.blocks_over_limit") + ( + ("route", "spend_reporting"), + ) + + def test_a_test_with_only_the_old_string_covers_is_unchanged(self, request: pytest.FixtureRequest) -> None: + test = type(self).test_a_test_with_only_the_old_string_covers_is_unchanged + request.applymarker(pytest.mark.covers("llm.responses.openai.tool_use.nonstream.works")) + item = collected_item(request, test.__name__) + assert result_properties(item) == fixed_prefix(item, "llm.responses.openai.tool_use.nonstream.works") + + def test_a_test_with_neither_marker_carries_only_the_prefix(self, request: pytest.FixtureRequest) -> None: + test = type(self).test_a_test_with_neither_marker_carries_only_the_prefix + item = collected_item(request, test.__name__) + assert subject_properties(item) == () + assert result_properties(item) == fixed_prefix(item, "") + + def test_a_marker_carrying_something_other_than_a_subject_emits_nothing( + self, request: pytest.FixtureRequest + ) -> None: + test = type(self).test_a_marker_carrying_something_other_than_a_subject_emits_nothing + request.applymarker(pytest.mark.meta("spend-budgets")) + assert subject_properties(collected_item(request, test.__name__)) == () + + +class TestProviderMirrorsLitellm: + """`Provider` copies `LlmProviders` values so collecting tests/e2e never needs litellm; skips where it is absent.""" + + def test_every_provider_value_is_a_real_litellm_provider(self) -> None: + try: + from litellm.types.utils import LlmProviders + except ImportError: # pragma: no cover - the runner image's shape + pytest.skip("litellm is not importable here, which is the property under test") + known = {str(member.value) for member in LlmProviders} + unknown = sorted(member.value for member in Provider if member.value not in known) + assert not unknown, f"not LlmProviders values: {unknown}" + + +E2E_DIR: Final = Path(__file__).resolve().parents[1] / "e2e" + + +def _hand_typed_models(path: Path) -> Iterator[str]: + for node in ast.walk(ast.parse(path.read_text())): + match node: + case ast.Call(func=ast.Name(id="Subject"), keywords=keywords): + for keyword in keywords: + match keyword: + case ast.keyword(arg="models", value=ast.Tuple(elts=models)): + yield from ( + f"{path.relative_to(E2E_DIR)}:{model.lineno} {model.value!r}" + for model in models + if isinstance(model, ast.Constant) + ) + case _: + pass + case _: + pass + + +def test_a_declared_model_names_the_constant_the_test_drives() -> None: + offenders: Final = tuple( + offender for path in sorted(E2E_DIR.rglob("*.py")) for offender in _hand_typed_models(path) + ) + assert offenders == () + + class TestStepRecording: """`@step`-decorated harness helpers append to the running test's story as they execute. diff --git a/tests/e2e/AGENTS.md b/tests/e2e/AGENTS.md index cbecc1adee7..920ca8b02a3 100644 --- a/tests/e2e/AGENTS.md +++ b/tests/e2e/AGENTS.md @@ -131,6 +131,29 @@ Current limits: Bedrock cannot be mounted in record or replay (SigV4 signs the H The harness is fully typed with no error budget: `make lint-e2e-basedpyright` must report zero basedpyright errors, and CI enforces that on any PR touching `tests/e2e/**/*.py`. When a response field is untyped, model it in `models.py` (just the fields you read) and let pydantic validate it, rather than threading a `dict` or `Any` through the test +## Typed test metadata + +Separate from the coverage registry and additive to it: `@meta(Subject(...))` from `e2e_metadata.py` says what a test DRIVES, as closed enums rather than a string id. `@pytest.mark.covers("cell.id")` is untouched and keeps working exactly as before; the two markers coexist on the same test, and `@meta` always goes BELOW `@covers` so `Item.location` still anchors at the first decorator and every `source` deep link stays put + +```python +@pytest.mark.covers("quota_management.budget.key.blocks_over_limit") +@meta( + Subject( + domain=Domain.SPEND_BUDGETS, + providers=(Provider.ANTHROPIC,), + models=(CHEAP_ANTHROPIC_MODEL,), + mode=Mode.NONSTREAM, + ) +) +def test_bare_key_blocks_over_its_own_budget(...) -> None: ... +``` + +`route` is the endpoint the test is checking: `TEAM_MANAGEMENT` for a `/team/update` test, `SPEND_REPORTING` for a `/spend/logs` test, `MESSAGES` for a test of spend on `/v1/messages`. A test whose chat call only triggers the behavior under test, like the budget block above, leaves it unset, since its steps already name the call + +Every field is optional today (the backfill of the rest of the suite is a later PR) and every field is a closed enum, so a typo is a basedpyright error at the call site rather than a property that silently never appears. `providers`, `models` and `capabilities` are tuples even with one member, because one test node routinely drives several: the claude_code matrix runs haiku, sonnet and opus in a single body, and a spend test calls two providers on one key. Declare every provider and every model the test drives, fallbacks included. The three are independent sets with no positional pairing between them (one provider x three models is the common case), and each is deduped and sorted at declaration so the committed run artifacts diff cleanly. `models=("gpt-5.5")` is a str and not a tuple, so anything but a tuple raises a `TypeError` where the decorator runs and shows up as a collection error naming the file. `Subject` is serialized with `dataclasses.asdict`, so a new scalar field needs no serializer edit; empty fields emit no `` at all. A declared model names the constant the test drives (`CHEAP_ANTHROPIC_MODEL`, the file's own `BACKEND`), never a copy of its value, so the property cannot claim one model while an env override runs another. `e2e_metadata` and its call sites never import litellm, only the stdlib, pytest and pydantic: `Provider` mirrors litellm's `LlmProviders` values instead of importing them, because tests/e2e is shipped to the runner image on its own and a `from litellm...` at module scope would make the litellm package a hard dependency of COLLECTING the suite. `TestProviderMirrorsLitellm` in `tests/code_coverage_tests/test_e2e_metadata.py` fails on drift wherever litellm is importable and skips where it is not, so adding a provider is one line in `e2e_metadata` + +Declared fields ride out as JUnit `` entries behind the fixed prefix, the same way steps do: each scalar under its field name, and each plural value as a repeated property under its SINGULAR name (`provider`, `model`, `capability`). The results JSON downstream regroups them under the plural key, so `providers`, `models` and `capabilities` are arrays there, `[]` when empty + ## Recorded test steps `@step` from `e2e_metadata.py` goes on harness helpers (client methods and poll loops), never on a test. Each call adds one plain-English sentence to the running test's list of steps, in call order, so the list reads as what the test did. The step is recorded before the helper runs, so when a test fails, its last step is where it failed. Nobody writes steps by hand. They come from the calls the test actually made, so they can't drift from what happened diff --git a/tests/e2e/conftest.py b/tests/e2e/conftest.py index 1995909efba..62153e38a83 100644 --- a/tests/e2e/conftest.py +++ b/tests/e2e/conftest.py @@ -121,6 +121,11 @@ def pytest_configure(config: pytest.Config) -> None: "markers", "covers(cell_id, *, exercised_on=()): coverage-registry cell(s) this test covers", ) + config.addinivalue_line( + "markers", + "meta(subject): typed e2e_metadata.Subject describing what this test drives" + " (domain/route/providers/models/capabilities/mode); attach it with @meta(Subject(...))", + ) config.addinivalue_line( "markers", "replayable: edge-wired test whose provider traffic replays from a fixture bundle, so it makes " diff --git a/tests/e2e/e2e_metadata.py b/tests/e2e/e2e_metadata.py index e5cd016e9d2..dc34b3db05b 100644 --- a/tests/e2e/e2e_metadata.py +++ b/tests/e2e/e2e_metadata.py @@ -1,14 +1,4 @@ -"""Per-test metadata for the e2e suite: the step log each test records as it runs. - -`steps` is appended at runtime by `@step`-decorated harness helpers, in call -order, so the list IS the test's user story and its last element is where a -failing test died. Nothing about it is hand-written, so it cannot drift from -what the test actually did. - -tests/e2e is a black-box HTTP suite that imports litellm in zero files and is -shipped to the runner image as tests/e2e alone, and every harness module imports -this one, so it imports only the stdlib and pydantic. -""" +"""Typed per-test metadata for the e2e suite: what a test drives (`Subject`) and what it did (`steps`). See AGENTS.md""" from __future__ import annotations @@ -20,13 +10,189 @@ import threading from collections import deque from collections.abc import Callable, Generator, Iterable, Mapping from contextlib import AbstractContextManager, contextmanager +from dataclasses import asdict, dataclass from enum import Enum from functools import reduce, wraps -from types import TracebackType +from itertools import chain +from types import MappingProxyType, TracebackType from typing import Final, ParamSpec, TypeVar, cast +import pytest from pydantic import BaseModel + +class Domain(str, Enum): + """The OSS issue-label taxonomy, so an issue and a test join on one string""" + + LLM_TRANSLATION = "llm-translation" + SPEND_BUDGETS = "spend-budgets" + UI = "ui" + MCP = "mcp" + OBSERVABILITY = "observability" + ROUTING = "routing" + DEPLOY_OPS = "deploy-ops" + COST_MAP = "cost-map" + PROXY_AUTH = "proxy-auth" + GUARDRAILS = "guardrails" + MANAGEMENT = "management" + SDK = "sdk" + PASSTHROUGH = "passthrough" + DB = "db" + CACHING = "caching" + DOCS = "docs" + AGENTS_API = "agents-api" + UNKNOWN = "unknown" + + +class Route(str, Enum): + """The endpoint the test is checking; unset when the call only triggers the behavior under test""" + + CHAT_COMPLETIONS = "chat_completions" + MESSAGES = "messages" + RESPONSES = "responses" + EMBEDDINGS = "embeddings" + COMPLETIONS = "completions" + FILES = "files" + BATCHES = "batches" + PASSTHROUGH = "passthrough" + MCP = "mcp" + GUARDRAILS = "guardrails" + KEY_MANAGEMENT = "key_management" + TEAM_MANAGEMENT = "team_management" + SPEND_REPORTING = "spend_reporting" + MODEL_MANAGEMENT = "model_management" + IMAGES = "images" + AUDIO = "audio" + MODERATIONS = "moderations" + RERANK = "rerank" + OCR = "ocr" + VECTOR_STORES = "vector_stores" + REALTIME = "realtime" + A2A = "a2a" + USER_MANAGEMENT = "user_management" + BUDGET_MANAGEMENT = "budget_management" + ORGANIZATION_MANAGEMENT = "organization_management" + CUSTOMER_MANAGEMENT = "customer_management" + HEALTH = "health" + METRICS = "metrics" + PROXY_CONFIG = "proxy_config" + ADMIN_UI = "admin_ui" + + +class Provider(str, Enum): + """Mirrors litellm's `LlmProviders` without importing litellm; `TestProviderMirrorsLitellm` catches drift""" + + OPENAI = "openai" + OPENAI_LIKE = "openai_like" + CUSTOM_OPENAI = "custom_openai" + AZURE = "azure" + AZURE_AI = "azure_ai" + ANTHROPIC = "anthropic" + GEMINI = "gemini" + VERTEX_AI = "vertex_ai" + BEDROCK = "bedrock" + SAGEMAKER = "sagemaker" + XAI = "xai" + GROQ = "groq" + DEEPSEEK = "deepseek" + MISTRAL = "mistral" + COHERE = "cohere" + PERPLEXITY = "perplexity" + OPENROUTER = "openrouter" + TOGETHER_AI = "together_ai" + FIREWORKS_AI = "fireworks_ai" + CEREBRAS = "cerebras" + SAMBANOVA = "sambanova" + NVIDIA_NIM = "nvidia_nim" + DATABRICKS = "databricks" + WATSONX = "watsonx" + OLLAMA = "ollama" + VLLM = "vllm" + HOSTED_VLLM = "hosted_vllm" + VOYAGE = "voyage" + JINA_AI = "jina_ai" + DEEPGRAM = "deepgram" + ELEVENLABS = "elevenlabs" + ASSEMBLYAI = "assemblyai" + LITELLM_PROXY = "litellm_proxy" + + +class Capability(str, Enum): + """A model feature, 1:1 with a `supports_*` key in model_prices_and_context_window.json""" + + FUNCTION_CALLING = "function_calling" + PARALLEL_FUNCTION_CALLING = "parallel_function_calling" + TOOL_CHOICE = "tool_choice" + TOOL_SEARCH = "tool_search" + VISION = "vision" + PDF_INPUT = "pdf_input" + AUDIO_INPUT = "audio_input" + REASONING = "reasoning" + WEB_SEARCH = "web_search" + PROMPT_CACHING = "prompt_caching" + RESPONSE_SCHEMA = "response_schema" + MID_CONVERSATION_SYSTEM = "mid_conversation_system" + + +class Mode(str, Enum): + """How the route was driven""" + + NONSTREAM = "nonstream" + STREAM = "stream" + BATCH = "batch" + WEBSOCKET = "websocket" + + +_M = TypeVar("_M") + + +def _scalar(value: object) -> str: + """`str()` on a (str, Enum) gives `Route.RESPONSES`, and StrEnum needs 3.11""" + if isinstance(value, Enum): + return str(value.value) # pyright: ignore[reportAny] # Enum.value is Any for every enum + return str(value) + + +def _members(value: object) -> tuple[object, ...] | None: + return cast("tuple[object, ...]", value) if isinstance(value, tuple) else None + + +def _canonical(name: str, value: object, member_type: type[_M]) -> tuple[_M, ...]: + """Validated, deduped and sorted; a bare str like `("gpt-5.5")` raises at import""" + members = _members(value) + if members is None: + raise TypeError( + f"Subject.{name} must be a tuple, got {type(value).__name__}: {value!r}." + f" A one-member tuple needs its trailing comma: {name}=(x,), not {name}=(x)" + ) + typed = tuple(member for member in members if isinstance(member, member_type)) + if len(typed) != len(members): + raise TypeError(f"Subject.{name} takes {member_type.__name__} members, got {value!r}") + return tuple(sorted(frozenset(member for member in typed if _scalar(member)), key=_scalar)) + + +@dataclass(frozen=True, slots=True) +class Subject: + """What a test is about. Not named `Test*` so pytest does not try to collect it""" + + domain: Domain | None = None + route: Route | None = None + providers: tuple[Provider, ...] = () + models: tuple[str, ...] = () + capabilities: tuple[Capability, ...] = () + mode: Mode | None = None + + def __post_init__(self) -> None: + object.__setattr__(self, "providers", _canonical("providers", self.providers, Provider)) + object.__setattr__(self, "models", _canonical("models", self.models, str)) + object.__setattr__(self, "capabilities", _canonical("capabilities", self.capabilities, Capability)) + + +def meta(subject: Subject) -> pytest.MarkDecorator: + """Attach a `Subject` to a test: `@meta(Subject(route=Route.RESPONSES, ...))`""" + return pytest.mark.meta(subject) + + _P = ParamSpec("_P") _R = TypeVar("_R") _Y = TypeVar("_Y") @@ -279,6 +445,35 @@ def step(label: str) -> Callable[[Callable[_P, _R]], Callable[_P, _R]]: return decorate +_REPEATED: Final = MappingProxyType({"providers": "provider", "models": "model", "capabilities": "capability"}) + + +def _declared_subject(args: tuple[object, ...]) -> Subject | None: + first = args[0] if args else None + return first if isinstance(first, Subject) else None + + +def subject_properties(item: pytest.Item) -> tuple[tuple[str, str], ...]: + """The declared fields as pairs, plural fields repeated under their singular name""" + marker: Final = item.get_closest_marker("meta") + if marker is None: + return () + subject: Final = _declared_subject(marker.args) + if subject is None: + return () + declared: Final[dict[str, object]] = asdict(subject) + return tuple(chain.from_iterable(_field_properties(name, value) for name, value in declared.items())) + + +def _field_properties(name: str, value: object) -> tuple[tuple[str, str], ...]: + repeated: Final = _REPEATED.get(name) + if repeated is not None: + return tuple((repeated, _scalar(member)) for member in _members(value) or ()) + if value is None or value == "": + return () + return ((name, _scalar(value)),) + + def step_properties() -> tuple[tuple[str, str], ...]: """The step log as repeated `step` properties. Appended after the setup and call phases, never at collection.""" diff --git a/tests/e2e/junit_properties.py b/tests/e2e/junit_properties.py index 9ee1ceebc96..c598515c918 100644 --- a/tests/e2e/junit_properties.py +++ b/tests/e2e/junit_properties.py @@ -20,7 +20,7 @@ from collections.abc import Iterable import pytest from coverage_registry.management_cases import case_properties -from e2e_metadata import step_properties +from e2e_metadata import step_properties, subject_properties # Hardcoded because the runner image copies tests/e2e/ to /app/e2e, so nothing # at runtime names this suite's place in the repo. test_junit_properties.py @@ -89,14 +89,16 @@ def covers_from_item(item: pytest.Item) -> tuple[str, ...]: def result_properties(item: pytest.Item) -> tuple[tuple[str, str], ...]: - """The custom signals a standard reporter cannot derive: the normalized suite - package, the comma-joined coverage-registry cell ids this test covers, and the - repo-relative `path:line` its source sits at.""" - return ( + """The custom signals a standard reporter cannot derive. + + Loki, Grafana and tests/integration/conftest.py read the `package`/`covers`/`source` prefix, so it never moves + """ + fixed = ( ("package", package_from_nodeid(item.nodeid)), ("covers", ",".join(covers_from_item(item))), ("source", source_from_item(item)), - ) + case_properties(item.nodeid) + ) + return fixed + case_properties(item.nodeid) + subject_properties(item) def attach_result_properties(item: pytest.Item) -> None: diff --git a/tests/e2e/pytest.ini b/tests/e2e/pytest.ini index d01caeff3ea..e795ebe5721 100644 --- a/tests/e2e/pytest.ini +++ b/tests/e2e/pytest.ini @@ -5,6 +5,7 @@ addopts = --strict-markers --strict-config --reruns 1 --only-rerun "kind='network'" --only-rerun "status_code=5[0-9][0-9]" markers = e2e: live test that requires a running proxy and real provider keys + meta: typed e2e_metadata.Subject describing what this test drives (domain/route/providers/models/capabilities/mode); attach it with @meta(Subject(...)), never as a bare pytest.mark replayable: edge-wired test whose provider traffic replays from a fixture bundle, so it makes zero provider calls in replay mode; the record/replay CI lane selects it with -m replayable load: heavy throughput/load test; collected last so it never perturbs latency-sensitive suites weekly: real-provider anomaly load test that spends real money; deselected unless E2E_WEEKLY_ANOMALY is set diff --git a/tests/e2e/quota_management/budgets/test_budget_crud_e2e.py b/tests/e2e/quota_management/budgets/test_budget_crud_e2e.py index 5070ec89704..520de814c85 100644 --- a/tests/e2e/quota_management/budgets/test_budget_crud_e2e.py +++ b/tests/e2e/quota_management/budgets/test_budget_crud_e2e.py @@ -10,12 +10,19 @@ from datetime import datetime, timezone import pytest from budget_client import BudgetClient +from e2e_metadata import Domain, Route, Subject, meta from lifecycle import ResourceManager pytestmark = pytest.mark.e2e @pytest.mark.covers("mgmt.budget.new.persists") +@meta( + Subject( + domain=Domain.SPEND_BUDGETS, + route=Route.BUDGET_MANAGEMENT, + ) +) def test_budget_crud_roundtrip(client: BudgetClient, resources: ResourceManager) -> None: budget_id = client.create_budget(max_budget=12.5, soft_budget=10.0, budget_duration="30d") resources.defer(lambda: client.delete_budget(budget_id)) @@ -38,6 +45,12 @@ def test_budget_crud_roundtrip(client: BudgetClient, resources: ResourceManager) @pytest.mark.covers("mgmt.budget.delete.persists") +@meta( + Subject( + domain=Domain.SPEND_BUDGETS, + route=Route.BUDGET_MANAGEMENT, + ) +) def test_budget_delete_removes_it(client: BudgetClient, resources: ResourceManager) -> None: budget_id = client.create_budget(max_budget=1.0) resources.defer(lambda: client.delete_budget(budget_id)) @@ -45,6 +58,12 @@ def test_budget_delete_removes_it(client: BudgetClient, resources: ResourceManag assert not client.budget_info(budget_id), "budget still present after delete" +@meta( + Subject( + domain=Domain.SPEND_BUDGETS, + route=Route.KEY_MANAGEMENT, + ) +) def test_budget_duration_schedules_reset_on_key(client: BudgetClient, resources: ResourceManager) -> None: key = client.generate_key(max_budget=10.0, budget_duration="30d") resources.defer(lambda: client.delete_key(key)) diff --git a/tests/e2e/quota_management/budgets/test_budget_enforcement_e2e.py b/tests/e2e/quota_management/budgets/test_budget_enforcement_e2e.py index 8a9be1d1385..d1e17548194 100644 --- a/tests/e2e/quota_management/budgets/test_budget_enforcement_e2e.py +++ b/tests/e2e/quota_management/budgets/test_budget_enforcement_e2e.py @@ -19,16 +19,18 @@ import pytest from budget_client import BudgetClient, is_budget_block from e2e_config import unique_marker from e2e_http import StreamingResponse, require_successful_call +from e2e_metadata import Domain, Mode, Provider, Subject, meta from lifecycle import ResourceManager pytestmark = pytest.mark.e2e +MODEL = "claude-haiku-4-5" TINY_CAP = 3e-6 ROOMY_CAP = 100.0 def _chat(client: BudgetClient, key: str, *, user: str | None = None) -> StreamingResponse: - return client.chat(key, "claude-haiku-4-5", f"spend {unique_marker()}", max_tokens=16, user=user) + return client.chat(key, MODEL, f"spend {unique_marker()}", max_tokens=16, user=user) def _assert_budget_blocks(client: BudgetClient, key: str, *, user: str = "") -> StreamingResponse: @@ -56,6 +58,14 @@ def _assert_blocked_422(client: BudgetClient, key: str) -> StreamingResponse: class TestBudgetBlocksPerLevel: @pytest.mark.covers("quota_management.budget.key.blocks_over_limit") + @meta( + Subject( + domain=Domain.SPEND_BUDGETS, + providers=(Provider.ANTHROPIC,), + models=(MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_bare_key_blocks_over_its_own_budget(self, client: BudgetClient, resources: ResourceManager) -> None: key = client.generate_key(max_budget=TINY_CAP) resources.defer(lambda: client.delete_key(key)) @@ -63,6 +73,14 @@ class TestBudgetBlocksPerLevel: _assert_blocked_422(client, key) @pytest.mark.covers("quota_management.budget.team.blocks_over_limit") + @meta( + Subject( + domain=Domain.SPEND_BUDGETS, + providers=(Provider.ANTHROPIC,), + models=(MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_team_budget_blocks_every_team_key(self, client: BudgetClient, resources: ResourceManager) -> None: team_id = client.create_team(alias=f"e2e-budget-team-{unique_marker()}", max_budget=TINY_CAP) resources.defer(lambda: client.delete_team(team_id)) @@ -79,6 +97,14 @@ class TestBudgetBlocksPerLevel: ) @pytest.mark.covers("quota_management.budget.internal_user.blocks_over_limit") + @meta( + Subject( + domain=Domain.SPEND_BUDGETS, + providers=(Provider.ANTHROPIC,), + models=(MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_user_budget_enforced_across_their_personal_keys( self, client: BudgetClient, resources: ResourceManager ) -> None: @@ -113,18 +139,34 @@ class TestBudgetBlocksPerLevel: require_successful_call(team_result) @pytest.mark.covers("quota_management.budget.end_user.blocks_over_limit") + @meta( + Subject( + domain=Domain.SPEND_BUDGETS, + providers=(Provider.ANTHROPIC,), + models=(MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_end_user_budget_blocks_attributed_calls( self, client: BudgetClient, resources: ResourceManager ) -> None: customer = f"e2e-budget-cust-{unique_marker()}" client.create_customer(customer, max_budget=TINY_CAP) resources.defer(lambda: client.delete_customers([customer])) - key = client.generate_key(models=["claude-haiku-4-5"]) + key = client.generate_key(models=[MODEL]) resources.defer(lambda: client.delete_key(key)) _assert_budget_blocks(client, key, user=customer) @pytest.mark.covers("quota_management.budget.organization.blocks_over_limit") + @meta( + Subject( + domain=Domain.SPEND_BUDGETS, + providers=(Provider.ANTHROPIC,), + models=(MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_org_budget_blocks_keys_under_it(self, client: BudgetClient, resources: ResourceManager) -> None: org_id = client.create_org(max_budget=TINY_CAP, alias=f"e2e-budget-org-{unique_marker()}") resources.defer(lambda: client.delete_org(org_id)) @@ -139,6 +181,14 @@ class TestBudgetBlocksPerLevel: ) @pytest.mark.covers("quota_management.budget.team_member.blocks_over_limit") + @meta( + Subject( + domain=Domain.SPEND_BUDGETS, + providers=(Provider.ANTHROPIC,), + models=(MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_member_budget_blocks_without_touching_teammates( self, client: BudgetClient, resources: ResourceManager ) -> None: @@ -166,6 +216,14 @@ class TestKeyBudgetBlocksAcrossKeyKinds: the capped key is refused, proving nothing around the key was the blocker.""" @pytest.mark.covers("quota_management.budget.key.blocks_over_limit") + @meta( + Subject( + domain=Domain.SPEND_BUDGETS, + providers=(Provider.ANTHROPIC,), + models=(MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_personal_key_blocks_over_its_own_budget( self, client: BudgetClient, resources: ResourceManager ) -> None: @@ -180,6 +238,14 @@ class TestKeyBudgetBlocksAcrossKeyKinds: require_successful_call(_chat(client, control_key)) @pytest.mark.covers("quota_management.budget.key.blocks_over_limit") + @meta( + Subject( + domain=Domain.SPEND_BUDGETS, + providers=(Provider.ANTHROPIC,), + models=(MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_team_key_blocks_over_its_own_budget(self, client: BudgetClient, resources: ResourceManager) -> None: team_id = client.create_team(alias=f"e2e-key-cap-team-{unique_marker()}", max_budget=ROOMY_CAP) resources.defer(lambda: client.delete_team(team_id)) @@ -192,6 +258,14 @@ class TestKeyBudgetBlocksAcrossKeyKinds: require_successful_call(_chat(client, control_key)) @pytest.mark.covers("quota_management.budget.key.blocks_over_limit") + @meta( + Subject( + domain=Domain.SPEND_BUDGETS, + providers=(Provider.ANTHROPIC,), + models=(MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_team_member_key_blocks_over_its_own_budget( self, client: BudgetClient, resources: ResourceManager ) -> None: diff --git a/tests/e2e/quota_management/budgets/test_budget_fallback_e2e.py b/tests/e2e/quota_management/budgets/test_budget_fallback_e2e.py index fe6db8f0454..96fd999d836 100644 --- a/tests/e2e/quota_management/budgets/test_budget_fallback_e2e.py +++ b/tests/e2e/quota_management/budgets/test_budget_fallback_e2e.py @@ -10,6 +10,7 @@ import pytest from budget_client import BudgetClient, model_budget from e2e_config import unique_marker +from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta from lifecycle import ResourceManager from models import AnthropicMessagesResponse @@ -20,6 +21,15 @@ FALLBACK_MODEL = "gpt-5.5" @pytest.mark.covers("quota_management.budget.fallback.routes_to_fallback") +@meta( + Subject( + domain=Domain.SPEND_BUDGETS, + route=Route.MESSAGES, + providers=(Provider.ANTHROPIC, Provider.OPENAI), + models=(PRIMARY_MODEL, FALLBACK_MODEL), + mode=Mode.NONSTREAM, + ) +) def test_budget_fallback_reroutes_anthropic_messages_to_openai( client: BudgetClient, resources: ResourceManager ) -> None: diff --git a/tests/e2e/quota_management/budgets/test_budget_reset_advances_e2e.py b/tests/e2e/quota_management/budgets/test_budget_reset_advances_e2e.py index fdd868b6bac..57074ffbff4 100644 --- a/tests/e2e/quota_management/budgets/test_budget_reset_advances_e2e.py +++ b/tests/e2e/quota_management/budgets/test_budget_reset_advances_e2e.py @@ -22,11 +22,13 @@ import pytest from budget_client import BudgetClient, is_budget_block from e2e_config import unique_marker from e2e_http import require_successful_call +from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta from lifecycle import ResourceManager from models import BudgetWindow pytestmark = pytest.mark.e2e +MODEL = "claude-haiku-4-5" WINDOW_SECONDS = 30 RESET_DEADLINE_SECONDS = 150 TINY_CAP = 3e-6 @@ -34,7 +36,7 @@ SPEND_SETTLE_DEADLINE_SECONDS = 90 def _call(client: BudgetClient, key: str): - return client.chat(key, "claude-haiku-4-5", f"advance {unique_marker()}", max_tokens=16) + return client.chat(key, MODEL, f"advance {unique_marker()}", max_tokens=16) def _poll_key_spend(client: BudgetClient, key: str, settled: Callable[[float], bool], problem: str) -> None: @@ -70,6 +72,12 @@ def _drive_to_block(client: BudgetClient, key: str) -> None: # ---- Rung 1: scheduling exists at creation ----------------------------------- +@meta( + Subject( + domain=Domain.SPEND_BUDGETS, + route=Route.KEY_MANAGEMENT, + ) +) def test_key_with_budget_duration_schedules_reset_at_creation(client: BudgetClient, resources: ResourceManager) -> None: """Baseline: a key created with a budget_duration has budget_reset_at populated immediately. The reset job can only advance a timestamp that was scheduled in @@ -86,6 +94,14 @@ def test_key_with_budget_duration_schedules_reset_at_creation(client: BudgetClie @pytest.mark.covers("quota_management.budget.key.blocks_over_limit") +@meta( + Subject( + domain=Domain.SPEND_BUDGETS, + providers=(Provider.ANTHROPIC,), + models=(MODEL,), + mode=Mode.NONSTREAM, + ) +) def test_key_spend_blocks_at_cap(client: BudgetClient, resources: ResourceManager) -> None: """Sanity that the tiny cap is enforced before we test that it resets: spend accrues across calls and eventually returns budget_exceeded, never a 5xx.""" @@ -103,6 +119,14 @@ def test_key_spend_blocks_at_cap(client: BudgetClient, resources: ResourceManage @pytest.mark.covers("quota_management.budget.key.resets_after_window") +@meta( + Subject( + domain=Domain.SPEND_BUDGETS, + providers=(Provider.ANTHROPIC,), + models=(MODEL,), + mode=Mode.NONSTREAM, + ) +) def test_key_budget_reset_at_advances_after_window(client: BudgetClient, resources: ResourceManager) -> None: """The core #25109 guard: after the window elapses the reset job must move budget_reset_at strictly forward AND zero key.spend. The broken nullable-JSON @@ -139,6 +163,14 @@ def test_key_budget_reset_at_advances_after_window(client: BudgetClient, resourc @pytest.mark.covers("quota_management.budget.key_multi_window.resets_windows_independently") +@meta( + Subject( + domain=Domain.SPEND_BUDGETS, + providers=(Provider.ANTHROPIC,), + models=(MODEL,), + mode=Mode.NONSTREAM, + ) +) def test_multi_window_key_resets_each_window_independently(client: BudgetClient, resources: ResourceManager) -> None: """The JSON-backed path #25109 specifically touched. A tight 30s window and a roomy 1m window: the tight window must reset on its own boundary while the roomy @@ -183,6 +215,14 @@ def test_multi_window_key_resets_each_window_independently(client: BudgetClient, @pytest.mark.covers("quota_management.budget.team_member.resets_after_window") +@meta( + Subject( + domain=Domain.SPEND_BUDGETS, + providers=(Provider.ANTHROPIC,), + models=(MODEL,), + mode=Mode.NONSTREAM, + ) +) def test_team_member_budget_reset_at_advances(client: BudgetClient, resources: ResourceManager) -> None: """Per-team member windows are also JSON-backed. member_budget_reset_at must advance after the window; the explicit before None: """The other #25109 failure mode: a reset job that ERRORS on the nullable-JSON column surfaces to the caller as a non-budget 5xx. Across the whole reset wait diff --git a/tests/e2e/quota_management/budgets/test_budget_reset_e2e.py b/tests/e2e/quota_management/budgets/test_budget_reset_e2e.py index b7b7f269c47..016fa9037ca 100644 --- a/tests/e2e/quota_management/budgets/test_budget_reset_e2e.py +++ b/tests/e2e/quota_management/budgets/test_budget_reset_e2e.py @@ -7,10 +7,12 @@ import pytest from budget_client import BudgetClient, is_budget_block from e2e_config import unique_marker from e2e_http import require_successful_call +from e2e_metadata import Domain, Mode, Provider, Subject, meta from lifecycle import ResourceManager pytestmark = pytest.mark.e2e +MODEL = "claude-haiku-4-5" TINY_CAP = 3e-6 ROOMY_CAP = 100.0 WINDOW = "30s" @@ -18,7 +20,7 @@ RESET_DEADLINE_SECONDS = 150 def _call(client: BudgetClient, key: str): - return client.chat(key, "claude-haiku-4-5", f"reset {unique_marker()}", max_tokens=16) + return client.chat(key, MODEL, f"reset {unique_marker()}", max_tokens=16) def _drive_to_block(client: BudgetClient, key: str) -> None: @@ -49,6 +51,14 @@ def _poll_until_serves_again(client: BudgetClient, key: str) -> None: class TestBudgetResetPerLevel: @pytest.mark.covers("quota_management.budget.key.resets_after_window") + @meta( + Subject( + domain=Domain.SPEND_BUDGETS, + providers=(Provider.ANTHROPIC,), + models=(MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_bare_key_budget_resets_after_window(self, client: BudgetClient, resources: ResourceManager) -> None: key = client.generate_key(max_budget=TINY_CAP, budget_duration=WINDOW) resources.defer(lambda: client.delete_key(key)) @@ -57,6 +67,14 @@ class TestBudgetResetPerLevel: _poll_until_serves_again(client, key) @pytest.mark.covers("quota_management.budget.team.resets_after_window") + @meta( + Subject( + domain=Domain.SPEND_BUDGETS, + providers=(Provider.ANTHROPIC,), + models=(MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_team_budget_resets_after_window(self, client: BudgetClient, resources: ResourceManager) -> None: team_id = client.create_team( alias=f"e2e-team-reset-{unique_marker()}", max_budget=TINY_CAP, budget_duration=WINDOW @@ -69,6 +87,14 @@ class TestBudgetResetPerLevel: _poll_until_serves_again(client, key) @pytest.mark.covers("quota_management.budget.organization.resets_after_window") + @meta( + Subject( + domain=Domain.SPEND_BUDGETS, + providers=(Provider.ANTHROPIC,), + models=(MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_org_budget_resets_after_window(self, client: BudgetClient, resources: ResourceManager) -> None: org_id = client.create_org( max_budget=TINY_CAP, alias=f"e2e-org-reset-{unique_marker()}", budget_duration=WINDOW @@ -91,6 +117,14 @@ class TestBudgetResetPerLevel: _poll_until_serves_again(client, key) @pytest.mark.covers("quota_management.budget.internal_user.resets_after_window") + @meta( + Subject( + domain=Domain.SPEND_BUDGETS, + providers=(Provider.ANTHROPIC,), + models=(MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_personal_key_user_budget_resets_after_window( self, client: BudgetClient, resources: ResourceManager ) -> None: @@ -109,6 +143,14 @@ class TestKeyBudgetResetAcrossKeyKinds: the only thing that can block and the only thing that has to reset.""" @pytest.mark.covers("quota_management.budget.key.resets_after_window") + @meta( + Subject( + domain=Domain.SPEND_BUDGETS, + providers=(Provider.ANTHROPIC,), + models=(MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_personal_key_resets_after_window(self, client: BudgetClient, resources: ResourceManager) -> None: user_id = client.create_user(max_budget=ROOMY_CAP) resources.defer(lambda: client.delete_user(user_id)) @@ -119,6 +161,14 @@ class TestKeyBudgetResetAcrossKeyKinds: _poll_until_serves_again(client, key) @pytest.mark.covers("quota_management.budget.key.resets_after_window") + @meta( + Subject( + domain=Domain.SPEND_BUDGETS, + providers=(Provider.ANTHROPIC,), + models=(MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_team_key_resets_after_window(self, client: BudgetClient, resources: ResourceManager) -> None: team_id = client.create_team(alias=f"e2e-key-reset-team-{unique_marker()}", max_budget=ROOMY_CAP) resources.defer(lambda: client.delete_team(team_id)) @@ -129,6 +179,14 @@ class TestKeyBudgetResetAcrossKeyKinds: _poll_until_serves_again(client, key) @pytest.mark.covers("quota_management.budget.key.resets_after_window") + @meta( + Subject( + domain=Domain.SPEND_BUDGETS, + providers=(Provider.ANTHROPIC,), + models=(MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_team_member_key_resets_after_window(self, client: BudgetClient, resources: ResourceManager) -> None: team_id = client.create_team(alias=f"e2e-key-reset-team-{unique_marker()}", max_budget=ROOMY_CAP) resources.defer(lambda: client.delete_team(team_id)) diff --git a/tests/e2e/quota_management/budgets/test_model_access_group_budget_e2e.py b/tests/e2e/quota_management/budgets/test_model_access_group_budget_e2e.py index 9c927a31216..50c6fda7981 100644 --- a/tests/e2e/quota_management/budgets/test_model_access_group_budget_e2e.py +++ b/tests/e2e/quota_management/budgets/test_model_access_group_budget_e2e.py @@ -21,6 +21,7 @@ import pytest from budget_client import BudgetClient, is_budget_block from e2e_config import unique_marker from e2e_http import StreamingResponse, require_successful_call +from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta from lifecycle import ResourceManager from models import KeyGenerateBody, LiteLLMParamsBody, ModelInfoBody, ModelNewBody @@ -102,6 +103,14 @@ def drained(client: BudgetClient) -> Iterator[DrainedPool]: class TestModelAccessGroupBudget: @pytest.mark.covers("quota_management.budget.model_access_group.blocks_over_limit") + @meta( + Subject( + domain=Domain.SPEND_BUDGETS, + providers=(Provider.OPENAI,), + models=(BACKEND,), + mode=Mode.NONSTREAM, + ) + ) def test_the_key_that_drained_the_pool_stays_blocked( self, client: BudgetClient, drained: DrainedPool ) -> None: @@ -114,6 +123,14 @@ class TestModelAccessGroupBudget: ) @pytest.mark.covers("quota_management.budget.model_access_group.enforced_across_keys") + @meta( + Subject( + domain=Domain.SPEND_BUDGETS, + providers=(Provider.OPENAI,), + models=(BACKEND,), + mode=Mode.NONSTREAM, + ) + ) def test_a_key_that_spent_nothing_is_blocked_by_the_shared_pool( self, client: BudgetClient, resources: ResourceManager, drained: DrainedPool ) -> None: @@ -127,6 +144,14 @@ class TestModelAccessGroupBudget: ) @pytest.mark.covers("quota_management.budget.model_access_group.isolates_per_group") + @meta( + Subject( + domain=Domain.SPEND_BUDGETS, + providers=(Provider.OPENAI,), + models=(BACKEND,), + mode=Mode.NONSTREAM, + ) + ) def test_a_drained_group_does_not_block_a_different_group( self, client: BudgetClient, resources: ResourceManager, drained: DrainedPool ) -> None: @@ -141,6 +166,14 @@ class TestModelAccessGroupBudget: require_successful_call(result) @pytest.mark.covers("quota_management.budget.model_access_group.reports_spend") + @meta( + Subject( + domain=Domain.SPEND_BUDGETS, + route=Route.BUDGET_MANAGEMENT, + providers=(Provider.OPENAI,), + models=(BACKEND,), + ) + ) def test_the_budget_read_reports_the_spend_drawn_against_the_pool( self, client: BudgetClient, drained: DrainedPool ) -> None: diff --git a/tests/e2e/quota_management/budgets/test_model_max_budget_e2e.py b/tests/e2e/quota_management/budgets/test_model_max_budget_e2e.py index 87ff9d56ab2..c69b0e232ff 100644 --- a/tests/e2e/quota_management/budgets/test_model_max_budget_e2e.py +++ b/tests/e2e/quota_management/budgets/test_model_max_budget_e2e.py @@ -13,6 +13,7 @@ import pytest from budget_client import BudgetClient, is_budget_block, model_budget from e2e_config import unique_marker from e2e_http import require_successful_call +from e2e_metadata import Domain, Mode, Provider, Subject, meta from lifecycle import ResourceManager from models import ModelBudgetEntry @@ -30,6 +31,14 @@ def _call(client: BudgetClient, key: str, model: str): @pytest.mark.covers("quota_management.budget.model_max.isolates_per_model") +@meta( + Subject( + domain=Domain.SPEND_BUDGETS, + providers=(Provider.ANTHROPIC, Provider.GEMINI), + models=(CAPPED_MODEL, FREE_MODEL), + mode=Mode.NONSTREAM, + ) +) def test_model_max_budget_isolates_per_model( client: BudgetClient, resources: ResourceManager ) -> None: @@ -61,6 +70,14 @@ def test_model_max_budget_isolates_per_model( @pytest.mark.skip(reason="stage red: product gap, end-user model_max_budget rpm_limit is stored but never enforced") @pytest.mark.covers("quota_management.budget.end_user_model_max.blocks_over_limit") +@meta( + Subject( + domain=Domain.SPEND_BUDGETS, + providers=(Provider.GEMINI,), + models=(FREE_MODEL,), + mode=Mode.NONSTREAM, + ) +) def test_end_user_model_max_budget_enforces_per_model_rpm( client: BudgetClient, resources: ResourceManager ) -> None: diff --git a/tests/e2e/quota_management/budgets/test_multi_window_budget_e2e.py b/tests/e2e/quota_management/budgets/test_multi_window_budget_e2e.py index e04f857545d..ddbc71cda9f 100644 --- a/tests/e2e/quota_management/budgets/test_multi_window_budget_e2e.py +++ b/tests/e2e/quota_management/budgets/test_multi_window_budget_e2e.py @@ -22,6 +22,7 @@ import pytest from budget_client import BudgetClient, is_budget_block, window_reset_at from e2e_http import StreamingResponse, require_successful_call from e2e_config import CHEAP_OPENAI_MODEL, unique_marker +from e2e_metadata import Domain, Mode, Provider, Subject, meta from lifecycle import ResourceManager from models import BudgetWindow @@ -57,6 +58,14 @@ def _drive_to_block(client: BudgetClient, key: str) -> StreamingResponse: @pytest.mark.covers("quota_management.budget.key_multi_window.blocks_then_resets") +@meta( + Subject( + domain=Domain.SPEND_BUDGETS, + providers=(Provider.OPENAI,), + models=(CHEAP_OPENAI_MODEL,), + mode=Mode.NONSTREAM, + ) +) def test_short_window_blocks_then_resets(client: BudgetClient, resources: ResourceManager) -> None: key = client.generate_key( models=[MODEL], @@ -90,6 +99,14 @@ def test_short_window_blocks_then_resets(client: BudgetClient, resources: Resour @pytest.mark.covers("quota_management.budget.key_multi_window.blocks_then_resets") +@meta( + Subject( + domain=Domain.SPEND_BUDGETS, + providers=(Provider.OPENAI,), + models=(CHEAP_OPENAI_MODEL,), + mode=Mode.NONSTREAM, + ) +) def test_long_window_blocks_after_short_window_resets(client: BudgetClient, resources: ResourceManager) -> None: key = client.generate_key( models=[MODEL], diff --git a/tests/e2e/quota_management/budgets/test_soft_budget_e2e.py b/tests/e2e/quota_management/budgets/test_soft_budget_e2e.py index 2006efb5a57..f04f4af0a8f 100644 --- a/tests/e2e/quota_management/budgets/test_soft_budget_e2e.py +++ b/tests/e2e/quota_management/budgets/test_soft_budget_e2e.py @@ -12,12 +12,23 @@ import pytest from budget_client import BudgetClient, is_budget_block from e2e_config import unique_marker from e2e_http import require_successful_call +from e2e_metadata import Domain, Mode, Provider, Subject, meta from lifecycle import ResourceManager pytestmark = pytest.mark.e2e +MODEL = "claude-haiku-4-5" + @pytest.mark.covers("quota_management.budget.soft.alerts_without_blocking") +@meta( + Subject( + domain=Domain.SPEND_BUDGETS, + providers=(Provider.ANTHROPIC,), + models=(MODEL,), + mode=Mode.NONSTREAM, + ) +) def test_soft_budget_does_not_block( client: BudgetClient, resources: ResourceManager ) -> None: @@ -27,7 +38,7 @@ def test_soft_budget_does_not_block( for _ in range(3): result = client.chat( - key, "claude-haiku-4-5", f"hi {unique_marker()}", max_tokens=16 + key, MODEL, f"hi {unique_marker()}", max_tokens=16 ) assert not is_budget_block(result), ( "soft_budget blocked a request; it must alert only, not block " diff --git a/tests/e2e/quota_management/budgets/test_spend_counter_reseed_e2e.py b/tests/e2e/quota_management/budgets/test_spend_counter_reseed_e2e.py index 4a69135cdd1..efeeaf90969 100644 --- a/tests/e2e/quota_management/budgets/test_spend_counter_reseed_e2e.py +++ b/tests/e2e/quota_management/budgets/test_spend_counter_reseed_e2e.py @@ -28,6 +28,7 @@ from pydantic import TypeAdapter, ValidationError from budget_client import BudgetClient from e2e_config import unique_marker from e2e_http import StreamingResponse +from e2e_metadata import Domain, Mode, Provider, Subject, meta from lifecycle import ResourceManager if TYPE_CHECKING: @@ -144,6 +145,14 @@ def _accumulate(client: BudgetClient, key: str, count: int) -> None: @pytest.mark.covers("quota_management.budget.spend_counter.reseed_matches_db") +@meta( + Subject( + domain=Domain.SPEND_BUDGETS, + providers=(Provider.ANTHROPIC,), + models=(MODEL,), + mode=Mode.NONSTREAM, + ) +) def test_cold_counter_reseed_keeps_counter_equal_to_db_spend( client: BudgetClient, resources: ResourceManager ) -> None: diff --git a/tests/e2e/quota_management/budgets/test_tag_budget_e2e.py b/tests/e2e/quota_management/budgets/test_tag_budget_e2e.py index b0068c66630..1723250915c 100644 --- a/tests/e2e/quota_management/budgets/test_tag_budget_e2e.py +++ b/tests/e2e/quota_management/budgets/test_tag_budget_e2e.py @@ -13,17 +13,19 @@ import pytest from budget_client import BudgetClient, is_budget_block from e2e_config import unique_marker from e2e_http import require_successful_call +from e2e_metadata import Domain, Mode, Provider, Subject, meta from lifecycle import ResourceManager pytestmark = pytest.mark.e2e +MODEL = "claude-haiku-4-5" TINY_BUDGET = 1e-6 def _tagged_call(client: BudgetClient, key: str, tag: str): result = client.chat( key, - "claude-haiku-4-5", + MODEL, f"hi {unique_marker()}", tags=[tag], max_tokens=64, @@ -34,6 +36,14 @@ def _tagged_call(client: BudgetClient, key: str, tag: str): @pytest.mark.covers("quota_management.budget.tag.blocks_over_limit") +@meta( + Subject( + domain=Domain.SPEND_BUDGETS, + providers=(Provider.ANTHROPIC,), + models=(MODEL,), + mode=Mode.NONSTREAM, + ) +) def test_tag_budget_blocks_tagged_requests( client: BudgetClient, scoped_key: str, resources: ResourceManager ) -> None: diff --git a/tests/e2e/quota_management/budgets/test_team_member_budget_e2e.py b/tests/e2e/quota_management/budgets/test_team_member_budget_e2e.py index 0fd0a545660..a323342d66d 100644 --- a/tests/e2e/quota_management/budgets/test_team_member_budget_e2e.py +++ b/tests/e2e/quota_management/budgets/test_team_member_budget_e2e.py @@ -21,6 +21,7 @@ import pytest from budget_client import BudgetClient, is_budget_block from e2e_config import unique_marker from e2e_http import Success, require_successful_call +from e2e_metadata import Domain, Mode, Provider, Subject, meta from lifecycle import ResourceManager from models import ChatBody, ChatMessage @@ -79,6 +80,14 @@ def _send(client: BudgetClient, key: str) -> str | None: class TestTeamMemberBudget: + @meta( + Subject( + domain=Domain.SPEND_BUDGETS, + providers=(Provider.ANTHROPIC,), + models=(MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_member_spend_attributed_to_team_and_user(self, client: BudgetClient, member: _Member) -> None: sent = frozenset(rid for rid in (_send(client, member.key) for _ in range(BURST)) if rid) assert sent, "no member call went through; cannot check attribution" @@ -98,6 +107,14 @@ class TestTeamMemberBudget: ) @pytest.mark.covers("quota_management.budget.team_member.blocks_over_limit") + @meta( + Subject( + domain=Domain.SPEND_BUDGETS, + providers=(Provider.ANTHROPIC,), + models=(MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_member_spend_over_budget_is_blocked(self, client: BudgetClient, member: _Member) -> None: for _ in range(40): result = client.chat(member.key, MODEL, f"spend {unique_marker()}", max_tokens=16) diff --git a/tests/e2e/quota_management/budgets/test_team_member_budget_isolation_e2e.py b/tests/e2e/quota_management/budgets/test_team_member_budget_isolation_e2e.py index f03518f8a17..3a91b080db6 100644 --- a/tests/e2e/quota_management/budgets/test_team_member_budget_isolation_e2e.py +++ b/tests/e2e/quota_management/budgets/test_team_member_budget_isolation_e2e.py @@ -17,6 +17,7 @@ import pytest from budget_client import BudgetClient, is_budget_block from e2e_config import unique_marker from e2e_http import Success, require_successful_call +from e2e_metadata import Domain, Mode, Provider, Subject, meta from lifecycle import ResourceManager from models import ChatBody, ChatMessage @@ -88,6 +89,14 @@ def _roomy_send(client: BudgetClient, key: str) -> str: class TestTeamMemberBudgetIsolation: @pytest.mark.covers("quota_management.budget.team_member.isolates_per_member") + @meta( + Subject( + domain=Domain.SPEND_BUDGETS, + providers=(Provider.OPENAI,), + models=(MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_blocked_member_does_not_block_peer(self, client: BudgetClient, pair: _Pair) -> None: blocked = False for _ in range(40): diff --git a/tests/e2e/quota_management/budgets/test_team_member_budget_reset_e2e.py b/tests/e2e/quota_management/budgets/test_team_member_budget_reset_e2e.py index 5d097a81f92..2238006e869 100644 --- a/tests/e2e/quota_management/budgets/test_team_member_budget_reset_e2e.py +++ b/tests/e2e/quota_management/budgets/test_team_member_budget_reset_e2e.py @@ -6,10 +6,12 @@ import pytest from budget_client import BudgetClient from e2e_config import unique_marker from e2e_http import require_successful_call +from e2e_metadata import Domain, Mode, Provider, Subject, meta from lifecycle import ResourceManager pytestmark = pytest.mark.e2e +MODEL = "claude-haiku-4-5" MEMBER_BUDGET = 1.0 # default member budget is $50, we're testing with a smaller value def _as_datetime(value: str) -> datetime: @@ -17,6 +19,14 @@ def _as_datetime(value: str) -> datetime: @pytest.mark.covers("quota_management.budget.team_member.resets_after_window") +@meta( + Subject( + domain=Domain.SPEND_BUDGETS, + providers=(Provider.ANTHROPIC,), + models=(MODEL,), + mode=Mode.NONSTREAM, + ) +) def test_team_member_budget_reset_keeps_advancing(client: BudgetClient, resources: ResourceManager) -> None: team_id = client.create_team(alias=f"e2e-member-reset-{unique_marker()}", max_budget=100.0) resources.defer(lambda: client.delete_team(team_id)) @@ -34,7 +44,7 @@ def test_team_member_budget_reset_keeps_advancing(client: BudgetClient, resource # the member can spend within the team while the window is live key = client.generate_key(team_id=team_id, user_id=user_id) resources.defer(lambda: client.delete_key(key)) - require_successful_call(client.chat(key, "claude-haiku-4-5", f"reset {unique_marker()}", max_tokens=16)) + require_successful_call(client.chat(key, MODEL, f"reset {unique_marker()}", max_tokens=16)) # once the window elapses the reset job must move budget_reset_at forward; a job # that skips the member's budget row (the #25109 regression) leaves it pinned at diff --git a/tests/e2e/quota_management/budgets/test_team_multi_window_budget_e2e.py b/tests/e2e/quota_management/budgets/test_team_multi_window_budget_e2e.py index 7683132776b..e7696638b62 100644 --- a/tests/e2e/quota_management/budgets/test_team_multi_window_budget_e2e.py +++ b/tests/e2e/quota_management/budgets/test_team_multi_window_budget_e2e.py @@ -24,11 +24,13 @@ import pytest from budget_client import BudgetClient, is_budget_block, window_reset_at from e2e_http import StreamingResponse, require_successful_call from e2e_config import unique_marker +from e2e_metadata import Domain, Mode, Provider, Subject, meta from lifecycle import ResourceManager from models import BudgetWindow pytestmark = pytest.mark.e2e +MODEL = "claude-haiku-4-5" WINDOW_SECONDS = 30 SHORT_WINDOW = f"{WINDOW_SECONDS}s" LONG_WINDOW = "1d" @@ -38,7 +40,7 @@ RESET_DEADLINE_SECONDS = 150 def _call(client: BudgetClient, key: str): - return client.chat(key, "claude-haiku-4-5", f"team-window {unique_marker()}", max_tokens=16) + return client.chat(key, MODEL, f"team-window {unique_marker()}", max_tokens=16) def _drive_to_block(client: BudgetClient, key: str) -> StreamingResponse: @@ -52,6 +54,14 @@ def _drive_to_block(client: BudgetClient, key: str) -> StreamingResponse: @pytest.mark.covers("quota_management.budget.team_multi_window.blocks_then_resets") +@meta( + Subject( + domain=Domain.SPEND_BUDGETS, + providers=(Provider.ANTHROPIC,), + models=(MODEL,), + mode=Mode.NONSTREAM, + ) +) def test_team_short_window_blocks_then_resets(client: BudgetClient, resources: ResourceManager) -> None: team_id = client.create_team( alias=f"e2e-team-window-{unique_marker()}", @@ -61,7 +71,7 @@ def test_team_short_window_blocks_then_resets(client: BudgetClient, resources: R ], ) resources.defer(lambda: client.delete_team(team_id)) - key = client.generate_key(team_id=team_id, models=["claude-haiku-4-5"]) + key = client.generate_key(team_id=team_id, models=[MODEL]) resources.defer(lambda: client.delete_key(key)) # 1. exhaust the tight window -> litellm returns budget_exceeded @@ -85,6 +95,14 @@ def test_team_short_window_blocks_then_resets(client: BudgetClient, resources: R @pytest.mark.covers("quota_management.budget.team_multi_window.blocks_then_resets") +@meta( + Subject( + domain=Domain.SPEND_BUDGETS, + providers=(Provider.ANTHROPIC,), + models=(MODEL,), + mode=Mode.NONSTREAM, + ) +) def test_team_long_window_blocks_after_short_window_resets(client: BudgetClient, resources: ResourceManager) -> None: # 0. key with a short budget window and a long budget window @@ -96,7 +114,7 @@ def test_team_long_window_blocks_after_short_window_resets(client: BudgetClient, ], ) resources.defer(lambda: client.delete_team(team_id)) - key = client.generate_key(team_id=team_id, models=["claude-haiku-4-5"]) + key = client.generate_key(team_id=team_id, models=[MODEL]) resources.defer(lambda: client.delete_key(key)) # 1. drive the key to being blocked, assert its blocked by budget budget_exceeded diff --git a/tests/e2e/quota_management/budgets/test_user_budget_across_keys_e2e.py b/tests/e2e/quota_management/budgets/test_user_budget_across_keys_e2e.py index 4dc7a2df647..fb541897514 100644 --- a/tests/e2e/quota_management/budgets/test_user_budget_across_keys_e2e.py +++ b/tests/e2e/quota_management/budgets/test_user_budget_across_keys_e2e.py @@ -15,6 +15,7 @@ import pytest from budget_client import BudgetClient, is_budget_block from e2e_config import unique_marker from e2e_http import StreamingResponse, require_successful_call +from e2e_metadata import Domain, Mode, Provider, Subject, meta from lifecycle import ResourceManager pytestmark = pytest.mark.e2e @@ -58,6 +59,14 @@ def _expect_prompt_block(client: BudgetClient, key: str, subject: str) -> None: class TestUserBudgetAcrossKeys: @pytest.mark.covers("quota_management.budget.internal_user.enforced_across_keys") + @meta( + Subject( + domain=Domain.SPEND_BUDGETS, + providers=(Provider.OPENAI,), + models=(MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_user_budget_blocks_a_second_key(self, client: BudgetClient, resources: ResourceManager) -> None: user_id = client.create_user(max_budget=TINY_CAP) resources.defer(lambda: client.delete_user(user_id)) diff --git a/tests/e2e/quota_management/ratelimit/test_dynamic_rate_limit_priority_e2e.py b/tests/e2e/quota_management/ratelimit/test_dynamic_rate_limit_priority_e2e.py index a7d548381c1..da759a95d5f 100644 --- a/tests/e2e/quota_management/ratelimit/test_dynamic_rate_limit_priority_e2e.py +++ b/tests/e2e/quota_management/ratelimit/test_dynamic_rate_limit_priority_e2e.py @@ -46,6 +46,7 @@ from pydantic import BaseModel, ConfigDict, ValidationError from e2e_config import unique_marker from e2e_http import StreamingResponse, require_successful_call +from e2e_metadata import Domain, Mode, Provider, Subject, meta from lifecycle import ResourceManager from models import KeyGenerateBody, KeyMetadata, LiteLLMParamsBody from quota_client import QuotaClient @@ -157,6 +158,14 @@ class TestDynamicRateLimitPriority: "quota_management.ratelimit.priority_generous.picks_under_tpm", exercised_on=["chat_completions"], ) + @meta( + Subject( + domain=Domain.SPEND_BUDGETS, + providers=(Provider.ANTHROPIC,), + models=(BACKEND,), + mode=Mode.NONSTREAM, + ) + ) def test_generous_mode_lets_priority_borrow_past_reservation( self, client: QuotaClient, resources: ResourceManager ) -> None: @@ -199,6 +208,14 @@ class TestDynamicRateLimitPriority: "quota_management.ratelimit.priority_strict.picks_under_tpm", exercised_on=["chat_completions"], ) + @meta( + Subject( + domain=Domain.SPEND_BUDGETS, + providers=(Provider.ANTHROPIC,), + models=(BACKEND,), + mode=Mode.NONSTREAM, + ) + ) def test_strict_mode_blocks_saturated_priority_but_serves_the_other( self, client: QuotaClient, resources: ResourceManager ) -> None: diff --git a/tests/e2e/quota_management/ratelimit/test_rate_limit_e2e.py b/tests/e2e/quota_management/ratelimit/test_rate_limit_e2e.py index 7d87686b06c..22c91cf0836 100644 --- a/tests/e2e/quota_management/ratelimit/test_rate_limit_e2e.py +++ b/tests/e2e/quota_management/ratelimit/test_rate_limit_e2e.py @@ -39,6 +39,7 @@ from pydantic import BaseModel, ConfigDict, ValidationError from e2e_config import CHEAP_ANTHROPIC_MODEL, unique_marker from e2e_http import StreamingResponse, require_successful_call +from e2e_metadata import Domain, Mode, Provider, Subject, meta from lifecycle import ResourceManager from models import KeyGenerateBody from quota_client import QuotaClient @@ -176,6 +177,14 @@ def _assert_rate_limited(outcome: StreamingResponse, limit_type: str) -> None: class TestKeyRateLimits: @pytest.mark.covers("quota_management.ratelimit.rpm.blocks_over_limit") + @meta( + Subject( + domain=Domain.SPEND_BUDGETS, + providers=(Provider.ANTHROPIC,), + models=(CHEAP_ANTHROPIC_MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_rpm_limit_blocks_over_limit(self, client: QuotaClient, resources: ResourceManager) -> None: key = _limited_key(client, resources, rpm_limit=3) info = client.proxy.key_info(key) @@ -188,6 +197,14 @@ class TestKeyRateLimits: _assert_rate_limited(_chat(client, key), "requests") @pytest.mark.covers("quota_management.ratelimit.tpm.blocks_over_limit") + @meta( + Subject( + domain=Domain.SPEND_BUDGETS, + providers=(Provider.ANTHROPIC,), + models=(CHEAP_ANTHROPIC_MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_tpm_limit_blocks_over_limit(self, client: QuotaClient, resources: ResourceManager) -> None: key = _limited_key(client, resources, tpm_limit=TPM_LIMIT) info = client.proxy.key_info(key) @@ -207,6 +224,14 @@ class TestKeyRateLimits: _assert_rate_limited(_chat(client, key), "tokens") @pytest.mark.covers("quota_management.ratelimit.rpm.resets_after_window") + @meta( + Subject( + domain=Domain.SPEND_BUDGETS, + providers=(Provider.ANTHROPIC,), + models=(CHEAP_ANTHROPIC_MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_rpm_limit_resets_after_window(self, client: QuotaClient, resources: ResourceManager) -> None: key = _limited_key(client, resources, rpm_limit=1) @@ -232,6 +257,14 @@ class TestKeyRateLimits: pytest.fail("a blocked key never recovered after the rate-limit window elapsed") @pytest.mark.covers("quota_management.ratelimit.rpm.headers_report_remaining") + @meta( + Subject( + domain=Domain.SPEND_BUDGETS, + providers=(Provider.ANTHROPIC,), + models=(CHEAP_ANTHROPIC_MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_headers_report_limit_and_remaining(self, client: QuotaClient, resources: ResourceManager) -> None: key = _limited_key(client, resources, rpm_limit=5, tpm_limit=100000) diff --git a/tests/e2e/quota_management/ratelimit/test_redis_backed_ratelimit_e2e.py b/tests/e2e/quota_management/ratelimit/test_redis_backed_ratelimit_e2e.py index a88f0ca546a..83983ed33d5 100644 --- a/tests/e2e/quota_management/ratelimit/test_redis_backed_ratelimit_e2e.py +++ b/tests/e2e/quota_management/ratelimit/test_redis_backed_ratelimit_e2e.py @@ -13,6 +13,7 @@ import pytest from e2e_config import unique_marker from e2e_http import require_successful_call +from e2e_metadata import Domain, Mode, Provider, Subject, meta from lifecycle import ResourceManager from models import KeyGenerateBody, LiteLLMParamsBody from quota_client import QuotaClient @@ -40,6 +41,14 @@ class TestRedisBackedRateLimit: "quota_management.ratelimit.redis_backed.blocks_over_limit", exercised_on=["chat_completions"], ) + @meta( + Subject( + domain=Domain.SPEND_BUDGETS, + providers=(Provider.ANTHROPIC,), + models=(BACKEND,), + mode=Mode.NONSTREAM, + ) + ) def test_rpm_limit_one_blocks_second_call( self, client: QuotaClient, resources: ResourceManager ) -> None: diff --git a/tests/e2e/quota_management/ratelimit/test_redis_circuit_breaker_e2e.py b/tests/e2e/quota_management/ratelimit/test_redis_circuit_breaker_e2e.py index 3e1bc662470..fe49961146d 100644 --- a/tests/e2e/quota_management/ratelimit/test_redis_circuit_breaker_e2e.py +++ b/tests/e2e/quota_management/ratelimit/test_redis_circuit_breaker_e2e.py @@ -15,6 +15,7 @@ import pytest from e2e_config import unique_marker from e2e_http import require_successful_call +from e2e_metadata import Domain, Mode, Provider, Subject, meta from lifecycle import ResourceManager from models import KeyGenerateBody, LiteLLMParamsBody from quota_client import QuotaClient @@ -45,6 +46,14 @@ class TestRedisCircuitBreakerPath: "reliability.circuit_breaker.redis.trips_then_recovers", exercised_on=["chat_completions"], ) + @meta( + Subject( + domain=Domain.SPEND_BUDGETS, + providers=(Provider.ANTHROPIC,), + models=(BACKEND,), + mode=Mode.NONSTREAM, + ) + ) def test_burst_rate_limit_does_not_freeze_fresh_key( self, client: QuotaClient, resources: ResourceManager ) -> None: diff --git a/tests/e2e/quota_management/ratelimit/test_tpm_excludes_cached_tokens_e2e.py b/tests/e2e/quota_management/ratelimit/test_tpm_excludes_cached_tokens_e2e.py index 33d869ee80e..697bfe91b14 100644 --- a/tests/e2e/quota_management/ratelimit/test_tpm_excludes_cached_tokens_e2e.py +++ b/tests/e2e/quota_management/ratelimit/test_tpm_excludes_cached_tokens_e2e.py @@ -24,6 +24,7 @@ from models import ( TextBlock, Usage, ) +from e2e_metadata import Capability, Domain, Mode, Provider, Subject, meta from quota_client import QuotaClient pytestmark = [pytest.mark.e2e, pytest.mark.provider_live] @@ -101,6 +102,15 @@ class TestTpmExcludesCachedTokens: "quota_management.ratelimit.tpm.excludes_cached_tokens", exercised_on=["chat_completions"], ) + @meta( + Subject( + domain=Domain.SPEND_BUDGETS, + providers=(Provider.ANTHROPIC,), + models=(ANTHROPIC_MODEL,), + capabilities=(Capability.PROMPT_CACHING,), + mode=Mode.NONSTREAM, + ) + ) def test_cache_hit_reduces_tpm_by_non_cached_only( self, client: QuotaClient, resources: ResourceManager ) -> None: diff --git a/tests/e2e/quota_management/spend_tracking/spend_reconciliation.py b/tests/e2e/quota_management/spend_tracking/spend_reconciliation.py index 26809874aed..f313325dbda 100644 --- a/tests/e2e/quota_management/spend_tracking/spend_reconciliation.py +++ b/tests/e2e/quota_management/spend_tracking/spend_reconciliation.py @@ -9,6 +9,7 @@ from lifecycle import ResourceManager from models import ChatBody, ChatMessage, ChatResponse, KeyGenerateBody, LiteLLMParamsBody, TeamNewBody from spend_e2e_client import SpendClient +BACKEND: Final = "openai/gpt-5.6-luna" INPUT_RATE: Final = 0.00004 OUTPUT_RATE: Final = 0.00008 @@ -38,7 +39,7 @@ def create_traffic(client: SpendClient, resources: ResourceManager) -> tuple[Tea model_id: Final = client.proxy.create_model( model, LiteLLMParamsBody( - model="openai/gpt-5.6-luna", + model=BACKEND, api_key="os.environ/OPENAI_API_KEY", api_base=None if base is None else f"{base}/v1", input_cost_per_token=INPUT_RATE, diff --git a/tests/e2e/quota_management/spend_tracking/test_cache_cost_accounting_e2e.py b/tests/e2e/quota_management/spend_tracking/test_cache_cost_accounting_e2e.py index c50ec3d902f..ff9710dca2b 100644 --- a/tests/e2e/quota_management/spend_tracking/test_cache_cost_accounting_e2e.py +++ b/tests/e2e/quota_management/spend_tracking/test_cache_cost_accounting_e2e.py @@ -52,6 +52,7 @@ from cost_rows import ( ) from e2e_config import unique_marker from e2e_http import unwrap +from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta from lifecycle import ResourceManager from models import AnthropicMessagesBody, ChatBody, ChatMessage, LiteLLMParamsBody from pydantic import BaseModel @@ -122,6 +123,15 @@ def _assert_cache_read_billed(row: CostRow) -> None: class TestCacheCostAccounting: @pytest.mark.covers("quota_management.spend_tracking.cache_write.bills_cache_creation_rate") + @meta( + Subject( + domain=Domain.SPEND_BUDGETS, + providers=(Provider.OPENAI,), + models=(CACHE_WRITE_BACKEND,), + capabilities=(Capability.PROMPT_CACHING,), + mode=Mode.NONSTREAM, + ) + ) def test_cache_write_tokens_billed_at_cache_creation_rate( self, client: SpendClient, resources: ResourceManager, scoped_key: str ) -> None: @@ -152,6 +162,15 @@ class TestCacheCostAccounting: assert_total_is_sum_of_components(row) @pytest.mark.covers("quota_management.spend_tracking.cost_breakdown.reports_component_costs") + @meta( + Subject( + domain=Domain.SPEND_BUDGETS, + providers=(Provider.OPENAI,), + models=(CACHE_READ_BACKEND,), + capabilities=(Capability.PROMPT_CACHING, Capability.REASONING), + mode=Mode.NONSTREAM, + ) + ) def test_cost_breakdown_reports_component_costs( self, client: SpendClient, resources: ResourceManager, scoped_key: str ) -> None: @@ -216,6 +235,15 @@ class TestCacheCostAccounting: _assert_cache_read_billed(row) @pytest.mark.covers("quota_management.spend_tracking.stream_cache_read.bills_cache_read_rate") + @meta( + Subject( + domain=Domain.SPEND_BUDGETS, + providers=(Provider.OPENAI,), + models=(CACHE_READ_BACKEND,), + capabilities=(Capability.PROMPT_CACHING,), + mode=Mode.STREAM, + ) + ) def test_streaming_cache_read_billed_at_cache_read_rate( self, client: SpendClient, resources: ResourceManager, scoped_key: str ) -> None: @@ -247,6 +275,16 @@ class TestCacheCostAccounting: _assert_cache_read_billed(row) @pytest.mark.covers("quota_management.spend_tracking.messages_bridge.keeps_cache_tokens") + @meta( + Subject( + domain=Domain.SPEND_BUDGETS, + route=Route.MESSAGES, + providers=(Provider.OPENAI,), + models=(BRIDGE_BACKEND,), + capabilities=(Capability.PROMPT_CACHING,), + mode=Mode.NONSTREAM, + ) + ) def test_messages_bridge_keeps_cache_tokens( self, client: SpendClient, resources: ResourceManager, scoped_key: str ) -> None: diff --git a/tests/e2e/quota_management/spend_tracking/test_cost_headers_e2e.py b/tests/e2e/quota_management/spend_tracking/test_cost_headers_e2e.py index abc321ccde8..0c4a4a87556 100644 --- a/tests/e2e/quota_management/spend_tracking/test_cost_headers_e2e.py +++ b/tests/e2e/quota_management/spend_tracking/test_cost_headers_e2e.py @@ -27,6 +27,7 @@ import pytest from cost_rows import approx_equal, cacheable_prefix, register_priced_model from e2e_config import unique_marker from e2e_http import StreamingResponse +from e2e_metadata import Domain, Mode, Provider, Subject, meta from lifecycle import ResourceManager from models import ChatBody, ChatMessage, ChatResponse, LiteLLMParamsBody from spend_e2e_client import SpendClient @@ -60,6 +61,14 @@ def _header_cost(response: StreamingResponse, name: str) -> float: class TestCostHeaders: @pytest.mark.covers("quota_management.spend_tracking.cost_headers.additive_components") + @meta( + Subject( + domain=Domain.SPEND_BUDGETS, + providers=(Provider.OPENAI,), + models=(BACKEND,), + mode=Mode.NONSTREAM, + ) + ) def test_component_cost_headers_sum_to_total( self, client: SpendClient, resources: ResourceManager, scoped_key: str ) -> None: diff --git a/tests/e2e/quota_management/spend_tracking/test_key_attribution_e2e.py b/tests/e2e/quota_management/spend_tracking/test_key_attribution_e2e.py index 4a2c23927c6..e0c19ea1b6b 100644 --- a/tests/e2e/quota_management/spend_tracking/test_key_attribution_e2e.py +++ b/tests/e2e/quota_management/spend_tracking/test_key_attribution_e2e.py @@ -36,6 +36,7 @@ from datetime import datetime, timedelta, timezone from typing import Final import pytest +from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta from models import KeyGenerateBody from proxy_client import Converged, await_converged from pydantic import BaseModel @@ -61,6 +62,7 @@ EMBED_MODEL: Final = "openai-text-embedding-3-small" BATCH_MODEL: Final = "openai-gpt-4o-mini" BATCH_BACKEND_MODEL: Final = "gpt-4o-mini" BATCH_PROVIDER: Final = "openai" +DRIVEN_MODELS: Final = (CHAT_MODEL, MESSAGES_MODEL, RESPONSES_MODEL, EMBED_MODEL, BATCH_MODEL) HEALTH_SERVICE_ACCOUNT: Final = "litellm-internal-health-check" BATCH_TERMINAL_STATUSES: Final = frozenset({"completed", "failed", "cancelled", "expired"}) FAILED_BATCH_POLL_SECONDS: Final = 120.0 @@ -281,6 +283,14 @@ class TestKeyAttribution: "rust_control_plane", ], ) + @meta( + Subject( + domain=Domain.SPEND_BUDGETS, + route=Route.SPEND_REPORTING, + providers=(Provider.GEMINI, Provider.ANTHROPIC, Provider.OPENAI), + models=DRIVEN_MODELS, + ) + ) def test_every_write_path_row_joins_the_key(self, client: SpendClient, driven: DrivenKey) -> None: assert tuple(path.name for path in driven.paths) == WRITE_PATHS found: Final = tuple((path, client.proxy.poll_logs_for_request_id(path.request_id)) for path in driven.paths) @@ -317,6 +327,14 @@ class TestKeyAttribution: "rust_control_plane", ], ) + @meta( + Subject( + domain=Domain.SPEND_BUDGETS, + route=Route.SPEND_REPORTING, + providers=(Provider.GEMINI, Provider.ANTHROPIC, Provider.OPENAI), + models=DRIVEN_MODELS, + ) + ) def test_spend_logs_by_key_return_every_row_with_the_alias(self, client: SpendClient, driven: DrivenKey) -> None: expected_ids: Final = frozenset(path.request_id for path in driven.paths) rows: Final = client.poll_logs_for_key( @@ -345,6 +363,14 @@ class TestKeyAttribution: "rust_control_plane", ], ) + @meta( + Subject( + domain=Domain.SPEND_BUDGETS, + route=Route.SPEND_REPORTING, + providers=(Provider.GEMINI, Provider.ANTHROPIC, Provider.OPENAI), + models=DRIVEN_MODELS, + ) + ) def test_user_daily_activity_reports_alias_and_email(self, client: SpendClient, driven: DrivenKey) -> None: breakdown: Final[DailyActivityKeyBreakdown | None] = client.poll_daily_activity_for_key( driven.identity.token, @@ -367,6 +393,14 @@ class TestKeyAttribution: "quota_management.spend_tracking.key_attribution.health_rows_keep_service_account", exercised_on=["chat_completions"], ) + @meta( + Subject( + domain=Domain.SPEND_BUDGETS, + route=Route.HEALTH, + providers=(Provider.GEMINI,), + models=(CHAT_MODEL,), + ) + ) def test_health_check_rows_keep_the_service_account_key(self, client: SpendClient) -> None: started_at: Final = datetime.now(timezone.utc) probe: Final = client.health(CHAT_MODEL) @@ -380,6 +414,15 @@ class TestKeyAttribution: "quota_management.spend_tracking.key_attribution.retrieve_batch_cost_joins_retrieving_key", exercised_on=["batches"], ) + @meta( + Subject( + domain=Domain.SPEND_BUDGETS, + route=Route.BATCHES, + providers=(Provider.OPENAI,), + models=(BATCH_MODEL,), + mode=Mode.BATCH, + ) + ) def test_terminal_batch_cost_row_joins_the_retrieving_key(self, client: SpendClient, driven: DrivenKey) -> None: provider_batch_id: Final = _provider_batch_id(_driven_batch_id(driven)) fetched: Final = _await_terminal_batch(client, driven.identity.key, provider_batch_id) diff --git a/tests/e2e/quota_management/spend_tracking/test_provider_edge_spend_e2e.py b/tests/e2e/quota_management/spend_tracking/test_provider_edge_spend_e2e.py index 4931af4222d..1aae4d98e4b 100644 --- a/tests/e2e/quota_management/spend_tracking/test_provider_edge_spend_e2e.py +++ b/tests/e2e/quota_management/spend_tracking/test_provider_edge_spend_e2e.py @@ -14,6 +14,7 @@ write path are all still under test with zero provider calls. import pytest from e2e_config import CHEAP_OPENAI_MODEL, provider_edge_base +from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta from lifecycle import ResourceManager from models import LiteLLMParamsBody from spend_e2e_client import SpendClient, unique_marker, unwrap @@ -22,6 +23,15 @@ pytestmark = [pytest.mark.e2e, pytest.mark.replayable] @pytest.mark.covers("quota_management.spend_tracking.chat_completions.logs_cost") +@meta( + Subject( + domain=Domain.SPEND_BUDGETS, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.OPENAI,), + models=(f"openai/{CHEAP_OPENAI_MODEL}",), + mode=Mode.NONSTREAM, + ) +) def test_edge_wired_chat_writes_nonzero_spend_row( client: SpendClient, resources: ResourceManager, scoped_key: str ) -> None: diff --git a/tests/e2e/quota_management/spend_tracking/test_service_tier_pricing_e2e.py b/tests/e2e/quota_management/spend_tracking/test_service_tier_pricing_e2e.py index bf68fb68a60..0e3a03360c6 100644 --- a/tests/e2e/quota_management/spend_tracking/test_service_tier_pricing_e2e.py +++ b/tests/e2e/quota_management/spend_tracking/test_service_tier_pricing_e2e.py @@ -35,6 +35,7 @@ from cost_rows import ( ) from e2e_config import CHEAP_OPENAI_MODEL, unique_marker from e2e_http import unwrap +from e2e_metadata import Capability, Domain, Mode, Provider, Subject, meta from lifecycle import ResourceManager from models import ( AnthropicMessagesBody, @@ -97,6 +98,15 @@ def _served_tier(chunks: list[_StreamChunk]) -> str: class TestServiceTierPricing: @pytest.mark.covers("quota_management.spend_tracking.service_tier.bills_tier_rates") + @meta( + Subject( + domain=Domain.SPEND_BUDGETS, + providers=(Provider.OPENAI,), + models=(BACKEND,), + capabilities=(Capability.REASONING,), + mode=Mode.NONSTREAM, + ) + ) def test_priority_tier_bills_priority_rates( self, client: SpendClient, resources: ResourceManager, scoped_key: str ) -> None: diff --git a/tests/e2e/quota_management/spend_tracking/test_spend_routes.py b/tests/e2e/quota_management/spend_tracking/test_spend_routes.py index c3697a31424..7b5db9ccd27 100644 --- a/tests/e2e/quota_management/spend_tracking/test_spend_routes.py +++ b/tests/e2e/quota_management/spend_tracking/test_spend_routes.py @@ -17,11 +17,13 @@ fast: no batch-write wait, no provider calls. """ from datetime import datetime, timedelta, timezone +from types import MappingProxyType from typing import Final import pytest from e2e_http import ProbeResult +from e2e_metadata import Domain, Route, Subject, meta from models import DateRangeParams from spend_e2e_client import SpendClient @@ -103,13 +105,38 @@ def _probe(client: SpendClient, route: str) -> ProbeResult: return client.probe(route, params=_date_range()) -@pytest.mark.parametrize("route", SPEND_ROUTES) +_LIST_ROUTES: Final = MappingProxyType( + { + "/key/list": Route.KEY_MANAGEMENT, + "/user/list": Route.USER_MANAGEMENT, + "/team/list": Route.TEAM_MANAGEMENT, + "/organization/list": Route.ORGANIZATION_MANAGEMENT, + "/customer/list": Route.CUSTOMER_MANAGEMENT, + } +) + +_ROUTE_CASES: Final = tuple( + pytest.param( + path, + marks=meta(Subject(domain=Domain.SPEND_BUDGETS, route=_LIST_ROUTES.get(path, Route.SPEND_REPORTING))), + ) + for path in SPEND_ROUTES +) + + +@pytest.mark.parametrize("route", _ROUTE_CASES) def test_spend_route_responsive(client: SpendClient, route: str) -> None: result = _probe(client, route) print(f"{route} -> {result.status_code}\n{result.body[:600]}") assert result.healthy, f"{route} -> {result.status_code}\n{result.body[:600]}" +@meta( + Subject( + domain=Domain.SPEND_BUDGETS, + route=Route.SPEND_REPORTING, + ) +) def test_schema_listed_spend_routes_are_responsive(client: SpendClient) -> None: """Probe any spend GET route the schema lists that isn't in SPEND_ROUTES.""" schema = client.openapi() diff --git a/tests/e2e/quota_management/spend_tracking/test_spend_tracking_e2e.py b/tests/e2e/quota_management/spend_tracking/test_spend_tracking_e2e.py index 6633396b538..a4c37c2df94 100644 --- a/tests/e2e/quota_management/spend_tracking/test_spend_tracking_e2e.py +++ b/tests/e2e/quota_management/spend_tracking/test_spend_tracking_e2e.py @@ -22,6 +22,7 @@ from typing import Final import pytest from e2e_http import RateLimitedError, Success +from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta from lifecycle import ResourceManager from models import KeyGenerateBody, LiteLLMParamsBody, SpendLogs, SpendLogsParams from spend_e2e_client import ( @@ -32,9 +33,16 @@ from spend_e2e_client import ( unique_marker, unwrap, ) +from spend_reconciliation import BACKEND as TRAFFIC_BACKEND pytestmark = pytest.mark.e2e +GEMINI_MODEL = "gemini-2.5-flash" +CLAUDE_MODEL = "claude-haiku-4-5" +CODEX_MODEL = "openai-responses-codex" +EMBEDDING_MODEL = "openai-text-embedding-3-small" +OPENAI_BACKEND = "openai/gpt-5.5" + def _approx_equal(actual: float, expected: float) -> bool: """Within 1% or 1e-9 absolute - spend math, not exact float identity.""" @@ -70,13 +78,22 @@ def _require_row( @pytest.mark.covers("quota_management.spend_tracking.chat_completions.logs_cost") +@meta( + Subject( + domain=Domain.SPEND_BUDGETS, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.GEMINI,), + models=(GEMINI_MODEL,), + mode=Mode.NONSTREAM, + ) +) def test_chat_completion_writes_nonzero_spend_row( client: SpendClient, scoped_key: str ) -> None: chat = unwrap( client.chat( scoped_key, - "gemini-2.5-flash", + GEMINI_MODEL, f"reply with one word {unique_marker()}", max_tokens=16, ) @@ -90,7 +107,7 @@ def test_chat_completion_writes_nonzero_spend_row( assert (row.spend or 0) > 0, f"chat row should cost > 0: {_summarize(rows)}" assert row.status == "success" assert row.cache_hit != "True", "fresh call must not be a cache hit" - assert "gemini-2.5-flash" in (row.model or "") + assert GEMINI_MODEL in (row.model or "") prompt = row.prompt_tokens or 0 completion = row.completion_tokens or 0 @@ -105,12 +122,21 @@ def test_chat_completion_writes_nonzero_spend_row( @pytest.mark.covers("quota_management.spend_tracking.stream.logs_cost") +@meta( + Subject( + domain=Domain.SPEND_BUDGETS, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.GEMINI,), + models=(GEMINI_MODEL,), + mode=Mode.STREAM, + ) +) def test_streaming_chat_completion_tracks_spend( client: SpendClient, scoped_key: str ) -> None: result = client.chat_stream( scoped_key, - "gemini-2.5-flash", + GEMINI_MODEL, f"count to three {unique_marker()}", max_tokens=64, ) @@ -133,6 +159,15 @@ def test_streaming_chat_completion_tracks_spend( @pytest.mark.covers("quota_management.spend_tracking.messages_bridge.logs_cost") +@meta( + Subject( + domain=Domain.SPEND_BUDGETS, + route=Route.MESSAGES, + providers=(Provider.OPENAI,), + models=(CODEX_MODEL,), + mode=Mode.STREAM, + ) +) def test_streaming_messages_via_responses_bridge_tracks_spend( client: SpendClient, scoped_key: str ) -> None: @@ -150,7 +185,7 @@ def test_streaming_messages_via_responses_bridge_tracks_spend( """ result = client.messages_stream( scoped_key, - "openai-responses-codex", + CODEX_MODEL, f"reply with exactly one word {unique_marker()}", max_tokens=64, ) @@ -203,13 +238,22 @@ def test_streaming_messages_via_responses_bridge_tracks_spend( @pytest.mark.covers("quota_management.spend_tracking.embeddings.logs_cost") @pytest.mark.covers("llm.embeddings.openai.basic.nonstream.cost_logged") +@meta( + Subject( + domain=Domain.SPEND_BUDGETS, + route=Route.EMBEDDINGS, + providers=(Provider.OPENAI,), + models=(EMBEDDING_MODEL,), + mode=Mode.NONSTREAM, + ) +) def test_embedding_writes_nonzero_spend_row( client: SpendClient, scoped_key: str ) -> None: _ = unwrap( client.embed( scoped_key, - "openai-text-embedding-3-small", + EMBEDDING_MODEL, f"vectorize this sentence {unique_marker()}", ) ) @@ -226,6 +270,14 @@ def test_embedding_writes_nonzero_spend_row( @pytest.mark.covers("quota_management.spend_tracking.cache_hit.zero_cost") +@meta( + Subject( + domain=Domain.SPEND_BUDGETS, + providers=(Provider.GEMINI,), + models=(GEMINI_MODEL,), + mode=Mode.NONSTREAM, + ) +) def test_cache_hit_is_zero_cost_and_suffixed( client: SpendClient, scoped_key: str ) -> None: @@ -234,8 +286,8 @@ def test_cache_hit_is_zero_cost_and_suffixed( # populated. The marker keeps each run isolated - a fixed prompt would persist # in the shared response cache across runs and make both calls hit (flaky). prompt = f"What is the capital of France? Answer in one word. {unique_marker()}" - _ = unwrap(client.chat(scoped_key, "gemini-2.5-flash", prompt, max_tokens=16, cache=None)) - _ = unwrap(client.chat(scoped_key, "gemini-2.5-flash", prompt, max_tokens=16, cache=None)) + _ = unwrap(client.chat(scoped_key, GEMINI_MODEL, prompt, max_tokens=16, cache=None)) + _ = unwrap(client.chat(scoped_key, GEMINI_MODEL, prompt, max_tokens=16, cache=None)) rows = client.poll_logs_for_key( scoped_key, @@ -262,12 +314,20 @@ def test_cache_hit_is_zero_cost_and_suffixed( @pytest.mark.covers("quota_management.spend_tracking.key_rollup.matches_sum_of_logs") +@meta( + Subject( + domain=Domain.SPEND_BUDGETS, + providers=(Provider.GEMINI,), + models=(GEMINI_MODEL,), + mode=Mode.NONSTREAM, + ) +) def test_key_spend_equals_sum_of_logs(client: SpendClient, scoped_key: str) -> None: for _ in range(2): _ = unwrap( client.chat( scoped_key, - "gemini-2.5-flash", + GEMINI_MODEL, f"say hi {unique_marker()}", max_tokens=16, ) @@ -290,6 +350,14 @@ def test_key_spend_equals_sum_of_logs(client: SpendClient, scoped_key: str) -> N @pytest.mark.replayable @pytest.mark.covers("quota_management.spend_tracking.concurrent_burst.loses_no_spend") +@meta( + Subject( + domain=Domain.SPEND_BUDGETS, + providers=(Provider.OPENAI,), + models=(TRAFFIC_BACKEND,), + mode=Mode.NONSTREAM, + ) +) def test_burst_of_concurrent_calls_loses_no_spend( client: SpendClient, resources: ResourceManager ) -> None: @@ -307,6 +375,15 @@ def test_burst_of_concurrent_calls_loses_no_spend( @pytest.mark.covers("quota_management.spend_tracking.pagination.keeps_total") +@meta( + Subject( + domain=Domain.SPEND_BUDGETS, + route=Route.SPEND_REPORTING, + providers=(Provider.GEMINI,), + models=(GEMINI_MODEL,), + mode=Mode.NONSTREAM, + ) +) def test_spend_logs_v2_pagination_caps_pages_and_keeps_total( client: SpendClient, scoped_key: str ) -> None: @@ -323,7 +400,7 @@ def test_spend_logs_v2_pagination_caps_pages_and_keeps_total( _ = unwrap( client.chat( scoped_key, - "gemini-2.5-flash", + GEMINI_MODEL, f"page fodder {unique_marker()}", max_tokens=16, ) @@ -360,11 +437,19 @@ def test_spend_logs_v2_pagination_caps_pages_and_keeps_total( @pytest.mark.covers("quota_management.spend_tracking.tags.attributes_spend") +@meta( + Subject( + domain=Domain.SPEND_BUDGETS, + providers=(Provider.GEMINI,), + models=(GEMINI_MODEL,), + mode=Mode.NONSTREAM, + ) +) def test_request_tags_round_trip(client: SpendClient, scoped_key: str) -> None: tag = f"e2e-spend-{unique_marker()}" _ = unwrap( client.chat( - scoped_key, "gemini-2.5-flash", "tagged request", tags=[tag], max_tokens=16 + scoped_key, GEMINI_MODEL, "tagged request", tags=[tag], max_tokens=16 ) ) @@ -377,6 +462,14 @@ def test_request_tags_round_trip(client: SpendClient, scoped_key: str) -> None: @pytest.mark.covers("quota_management.spend_tracking.tags.attributes_spend") +@meta( + Subject( + domain=Domain.SPEND_BUDGETS, + providers=(Provider.GEMINI,), + models=(GEMINI_MODEL,), + mode=Mode.NONSTREAM, + ) +) def test_tag_spend_matches_sum_of_tagged_logs( client: SpendClient, scoped_key: str ) -> None: @@ -387,7 +480,7 @@ def test_tag_spend_matches_sum_of_tagged_logs( _ = unwrap( client.chat( scoped_key, - "gemini-2.5-flash", + GEMINI_MODEL, f"hi {unique_marker()}", tags=[tag], max_tokens=16, @@ -415,12 +508,20 @@ def test_tag_spend_matches_sum_of_tagged_logs( @pytest.mark.covers("quota_management.spend_tracking.end_user.attributes_spend") +@meta( + Subject( + domain=Domain.SPEND_BUDGETS, + providers=(Provider.GEMINI,), + models=(GEMINI_MODEL,), + mode=Mode.NONSTREAM, + ) +) def test_end_user_spend_attributed_on_row( client: SpendClient, scoped_key: str, resources: ResourceManager ) -> None: customer = resources.customer(f"e2e-cust-{unique_marker()}") _ = unwrap( - client.chat(scoped_key, "gemini-2.5-flash", "hi", user=customer, max_tokens=16) + client.chat(scoped_key, GEMINI_MODEL, "hi", user=customer, max_tokens=16) ) rows = client.poll_logs_for_key( @@ -448,7 +549,7 @@ def test_end_user_header_attributes_responses_row( {"authorization": f"Bearer {scoped_key}", header: customer, "x-litellm-tags": tag} ) sent = client.send_responses_with_headers( - headers, "openai-responses-codex", f"one word {unique_marker()}" + headers, CODEX_MODEL, f"one word {unique_marker()}" ) assert sent.ok, f"/v1/responses failed with {sent.status_code}: {sent.body[:300]}" @@ -468,6 +569,14 @@ def test_end_user_header_attributes_responses_row( @pytest.mark.covers("quota_management.spend_tracking.per_model.writes_own_rows") +@meta( + Subject( + domain=Domain.SPEND_BUDGETS, + providers=(Provider.GEMINI, Provider.ANTHROPIC), + models=(GEMINI_MODEL, CLAUDE_MODEL), + mode=Mode.NONSTREAM, + ) +) def test_each_model_on_a_shared_key_gets_its_own_row( client: SpendClient, scoped_key: str ) -> None: @@ -478,27 +587,27 @@ def test_each_model_on_a_shared_key_gets_its_own_row( sibling deployment, or collapses both calls onto one request_id fails here.""" gemini = unwrap( client.chat( - scoped_key, "gemini-2.5-flash", f"one word {unique_marker()}", max_tokens=16 + scoped_key, GEMINI_MODEL, f"one word {unique_marker()}", max_tokens=16 ) ) claude = unwrap( client.chat( - scoped_key, "claude-haiku-4-5", f"one word {unique_marker()}", max_tokens=16 + scoped_key, CLAUDE_MODEL, f"one word {unique_marker()}", max_tokens=16 ) ) def both_models_costed(rows: list[SpendLogRow]) -> bool: costed = [r.model or "" for r in rows if (r.spend or 0) > 0] - return any("gemini-2.5-flash" in m for m in costed) and any( - "claude-haiku-4-5" in m for m in costed + return any(GEMINI_MODEL in m for m in costed) and any( + CLAUDE_MODEL in m for m in costed ) rows = client.poll_logs_for_key(scoped_key, min_rows=2, predicate=both_models_costed) gemini_row = _require_row( - rows, lambda r: "gemini-2.5-flash" in (r.model or ""), "for the gemini call" + rows, lambda r: GEMINI_MODEL in (r.model or ""), "for the gemini call" ) claude_row = _require_row( - rows, lambda r: "claude-haiku-4-5" in (r.model or ""), "for the claude call" + rows, lambda r: CLAUDE_MODEL in (r.model or ""), "for the claude call" ) assert (gemini_row.spend or 0) > 0, f"gemini row should cost > 0: {_summarize(rows)}" @@ -517,13 +626,21 @@ def test_each_model_on_a_shared_key_gets_its_own_row( @pytest.mark.covers("quota_management.spend_tracking.failure.writes_failure_row") +@meta( + Subject( + domain=Domain.SPEND_BUDGETS, + providers=(Provider.OPENAI,), + models=(OPENAI_BACKEND,), + mode=Mode.NONSTREAM, + ) +) def test_failure_call_writes_failure_status_row( client: SpendClient, resources: ResourceManager, scoped_key: str ) -> None: model = f"e2e-spend-failure-{unique_marker()}" model_id = client.proxy.create_model( model, - LiteLLMParamsBody(model="openai/gpt-5.5", api_key="sk-invalid-e2e-failure-row"), + LiteLLMParamsBody(model=OPENAI_BACKEND, api_key="sk-invalid-e2e-failure-row"), ) resources.defer(lambda: client.proxy.delete_model(model_id)) @@ -550,7 +667,7 @@ def test_failure_rows_share_normalized_error_across_provider_wording( carries the same stable normalized_error cluster key.""" marker = unique_marker() deployments: Final = ( - (f"e2e-norm-openai-{marker}", "openai/gpt-5.5"), + (f"e2e-norm-openai-{marker}", OPENAI_BACKEND), (f"e2e-norm-anthropic-{marker}", "anthropic/claude-haiku-4-5"), ) for name, provider_model in deployments: @@ -593,7 +710,7 @@ def test_pre_call_rejection_row_attributes_provider_and_model_id( can count it.""" model = f"e2e-spend-precall-{unique_marker()}" model_id = client.proxy.create_model( - model, LiteLLMParamsBody(model="openai/gpt-5.5", api_key="os.environ/OPENAI_API_KEY") + model, LiteLLMParamsBody(model=OPENAI_BACKEND, api_key="os.environ/OPENAI_API_KEY") ) resources.defer(lambda: client.proxy.delete_model(model_id)) key = client.proxy.generate_key(KeyGenerateBody(models=[model], rpm_limit=1)) @@ -624,9 +741,17 @@ def test_pre_call_rejection_row_attributes_provider_and_model_id( @pytest.mark.covers("quota_management.spend_tracking.spend_calculate.returns_cost") +@meta( + Subject( + domain=Domain.SPEND_BUDGETS, + route=Route.SPEND_REPORTING, + providers=(Provider.GEMINI,), + models=(GEMINI_MODEL,), + ) +) def test_spend_calculate_returns_nonzero_cost(client: SpendClient) -> None: cost = client.calculate_spend( - "gemini-2.5-flash", "estimate the cost of this request" + GEMINI_MODEL, "estimate the cost of this request" ) assert cost > 0, ( "/spend/calculate returned 0 for gemini-2.5-flash; " @@ -634,6 +759,15 @@ def test_spend_calculate_returns_nonzero_cost(client: SpendClient) -> None: ) +@meta( + Subject( + domain=Domain.SPEND_BUDGETS, + route=Route.SPEND_REPORTING, + providers=(Provider.GEMINI,), + models=(GEMINI_MODEL,), + mode=Mode.NONSTREAM, + ) +) def test_spend_logs_endpoint_returns_spend( client: SpendClient, scoped_key: str ) -> None: @@ -644,7 +778,7 @@ def test_spend_logs_endpoint_returns_spend( call's nonzero spend must surface before the deadline.""" unwrap( client.chat( - scoped_key, "gemini-2.5-flash", f"spend logs {unique_marker()}", max_tokens=16 + scoped_key, GEMINI_MODEL, f"spend logs {unique_marker()}", max_tokens=16 ) ) diff --git a/tests/e2e/quota_management/spend_tracking/test_team_daily_activity_e2e.py b/tests/e2e/quota_management/spend_tracking/test_team_daily_activity_e2e.py index ef635e59743..c86b55dc990 100644 --- a/tests/e2e/quota_management/spend_tracking/test_team_daily_activity_e2e.py +++ b/tests/e2e/quota_management/spend_tracking/test_team_daily_activity_e2e.py @@ -14,11 +14,12 @@ from typing import Final import pytest from e2e_http import ProbeResult +from e2e_metadata import Domain, Provider, Route, Subject, meta from lifecycle import ResourceManager from proxy_client import Converged, await_converged from pydantic import BaseModel from spend_e2e_client import SpendClient -from spend_reconciliation import TeamTraffic, assert_logs_match, create_traffic +from spend_reconciliation import BACKEND, TeamTraffic, assert_logs_match, create_traffic pytestmark = pytest.mark.e2e @@ -82,6 +83,14 @@ def _probe(client: SpendClient, params: BaseModel) -> ProbeResult: class TestTeamDailyActivity: @pytest.mark.replayable @pytest.mark.covers("mgmt.team.daily_activity.happy_path") + @meta( + Subject( + domain=Domain.SPEND_BUDGETS, + route=Route.SPEND_REPORTING, + providers=(Provider.OPENAI,), + models=(BACKEND,), + ) + ) def test_valid_date_range_returns_results_and_metadata( self, client: SpendClient, resources: ResourceManager ) -> None: @@ -199,6 +208,12 @@ class TestTeamDailyActivity: assert empty.metadata.total_failed_requests == 0 @pytest.mark.covers("mgmt.team.daily_activity.missing_start_date_rejected") + @meta( + Subject( + domain=Domain.SPEND_BUDGETS, + route=Route.SPEND_REPORTING, + ) + ) def test_missing_start_date_is_rejected(self, client: SpendClient) -> None: end = datetime.now(timezone.utc).date().isoformat() result = _probe(client, TeamDailyActivityParams(end_date=end, page=1)) @@ -207,6 +222,12 @@ class TestTeamDailyActivity: ) @pytest.mark.covers("mgmt.team.daily_activity.missing_end_date_rejected") + @meta( + Subject( + domain=Domain.SPEND_BUDGETS, + route=Route.SPEND_REPORTING, + ) + ) def test_missing_end_date_is_rejected(self, client: SpendClient) -> None: start = (datetime.now(timezone.utc).date() - timedelta(days=1)).isoformat() result = _probe(client, TeamDailyActivityParams(start_date=start, page=1))