test(e2e): typed per-test metadata for the e2e suite (#42044)

* feat(e2e): give e2e tests typed metadata for what they drive

@meta(Subject(domain, route, providers, models, capabilities, mode)) declares
what a test is about with closed enums, and each field lands in the JUnit report
as a property. The quota_management suites are the first to declare it.

* docs(e2e): say e2e_metadata avoids litellm, not that it is stdlib-only

It already imports pydantic and pytest, both of which the suite needs to collect. The rule that matters is no litellm import

* test(e2e): declare models through the constant each test drives

43 @meta declarations in quota_management typed the model name out again, so changing the call would leave the coverage report naming the old model. Each file now has one constant used by both, and a guard fails on any model written as a string literal in @meta

* refactor(e2e): set route only when the endpoint is what the test checks

A budget or rate-limit test whose chat call only triggers the block now leaves route unset, since its steps already name the call. Tests of an endpoint keep it: budget CRUD, key creation, spend reporting reads, and the per-endpoint spend tests for chat, messages, embeddings, batches and health. The two /spend/logs tests tagged chat_completions are now spend_reporting

* refactor(e2e): build the declared properties without mutating a list

subject_properties seeded a list and grew it with append and extend. It now flattens one tuple per field, and the plural-name table is a read-only mapping

* fix(e2e): tag each spend-route probe with the endpoint it checks

The breadth test gave all 33 probes spend_reporting, so /key/list, /user/list, /team/list, /organization/list and /customer/list counted as spend reporting. Each case now carries its own route, with organization and customer management added to Route
This commit is contained in:
ryan-crabbe-berri 2026-09-30 21:03:21 -07:00 • committed by GitHub
parent 321be01877
commit ae05f7d2c1
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
37 changed files with 1281 additions and 59 deletions

View file

@ -38,7 +38,7 @@ from collections.abc import Iterator
from pathlib import Path
import pytest
from e2e_metadata import step
from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta, step
FIRST_ATTEMPT_MADE = Path(__file__).with_name("first-attempt-made")
@ -100,6 +100,20 @@ def test_passes_on_the_rerun(key: None) -> None:
FIRST_ATTEMPT_MADE.touch()
chat(ok=not first_attempt)
poll_spend_logs()
@meta(
Subject(
domain=Domain.LLM_TRANSLATION,
route=Route.MESSAGES,
providers=(Provider.BEDROCK, Provider.ANTHROPIC),
models=("claude-sonnet-4-5", "claude-opus-4-7", "claude-haiku-4-5"),
capabilities=(Capability.VISION, Capability.FUNCTION_CALLING),
mode=Mode.STREAM,
)
)
def test_declares_two_providers_and_three_models() -> None:
assert Provider.BEDROCK.value == "bedrock"
"""
WIDE_FINALIZER_SUITE: Final = """
@ -183,6 +197,15 @@ def pytest_runtest_logreport(report: pytest.TestReport) -> None:
out.write(json.dumps([report.nodeid.split("::")[-1], steps]) + "\\n")
"""
BARE_STR_SUITE: Final = """
from e2e_metadata import Subject, meta
@meta(Subject(models=("gpt-5.5")))
def test_never_collected() -> None:
assert Subject is not None
"""
Properties = tuple[tuple[str, str], ...]
FailedReport: Final = TypeAdapter(tuple[str, tuple[str, ...]])
@ -278,7 +301,7 @@ def report(request: pytest.FixtureRequest, tmp_path_factory: pytest.TempPathFact
assert xml.exists(), f"the child run wrote no JUnit report:\n{child.stdout}\n{child.stderr}"
testsuite: Final = next(ElementTree.parse(xml).getroot().iter("testsuite"))
outcomes: Final = {name: testsuite.get(name) for name in ("tests", "failures", "errors", "skipped")}
assert outcomes == {"tests": "6", "failures": "1", "errors": "2", "skipped": "0"}, child.stdout
assert outcomes == {"tests": "7", "failures": "1", "errors": "2", "skipped": "0"}, child.stdout
return properties_by_test(testsuite)
@ -340,3 +363,33 @@ def test_a_failed_phase_s_own_report_carries_the_steps(tmp_path: Path) -> None:
"test_oauth_dies_on_consent": ("open the consent page",),
"test_plain_dies_on_consent": ("open the consent page",),
}, child.stdout
class TestDeclaredPropertiesReachTheReport:
def test_repeated_provider_model_and_capability_round_trip(self, report: Mapping[str, Properties]) -> None:
declared: Final = tuple(
(prop, value)
for prop, value in report["test_declares_two_providers_and_three_models"]
if prop not in {"package", "covers", "source"}
)
assert declared == (
("domain", "llm-translation"),
("route", "messages"),
("provider", "anthropic"),
("provider", "bedrock"),
("model", "claude-haiku-4-5"),
("model", "claude-opus-4-7"),
("model", "claude-sonnet-4-5"),
("capability", "function_calling"),
("capability", "vision"),
("mode", "stream"),
)
class TestBareStrIsACollectionError:
def test_a_str_where_a_tuple_belongs_fails_collection_and_names_the_fix(self, tmp_path: Path) -> None:
write_suite(tmp_path, {"test_bare_str.py": BARE_STR_SUITE})
child: Final = run_child_pytest(tmp_path)
assert child.returncode == pytest.ExitCode.INTERRUPTED, child.stdout
assert "Subject.models must be a tuple, got str: 'gpt-5.5'" in child.stdout
assert "models=(x,), not models=(x)" in child.stdout

View file

@ -1,4 +1,4 @@
"""The e2e step recorder's edge cases: label templates, dedupe, the cap, nesting, context managers.
"""The e2e test metadata: `@meta(Subject(...))` properties and the step recorder's edge cases.
Harness logic, so it lives here rather than under tests/e2e, which holds only
tests that drive a live proxy. The harness modules are imported off
@ -18,12 +18,30 @@ import threading
import warnings
from collections.abc import Callable, Generator, Iterator, Mapping
from contextlib import contextmanager
from dataclasses import fields, replace
from pathlib import Path
from types import UnionType
from typing import Final, cast, get_args, get_type_hints
import pytest
from e2e_metadata import MASK, MAX_STEPS, STEP_FRAMES, STEPS, StepRecorder, environment_secrets, step
from e2e_metadata import (
MASK,
MAX_STEPS,
STEP_FRAMES,
STEPS,
Capability,
Domain,
Mode,
Provider,
Route,
StepRecorder,
Subject,
environment_secrets,
meta,
step,
subject_properties,
)
from junit_properties import package_from_nodeid, result_properties, source_from_item
from proxy_client import ProxyClient
from pydantic import BaseModel, Field
from pydantic.fields import FieldInfo
@ -38,6 +56,191 @@ def empty_step_log() -> Generator[None]:
STEPS.reset()
def collected_item(request: pytest.FixtureRequest, name: str) -> pytest.Item:
return next(item for item in request.session.items if item.path == request.path and item.name == name)
def fixed_prefix(item: pytest.Item, covers: str) -> tuple[tuple[str, str], ...]:
"""Spelled out rather than taken from `result_properties`, so a change to either fails a test."""
return (
("package", package_from_nodeid(item.nodeid)),
("covers", covers),
("source", source_from_item(item)),
)
class TestSubjectProperties:
"""Markers go on via `request.applymarker` so the coverage registry's collect-only pass never sees them."""
def test_every_declared_field_becomes_a_property_in_field_order(self, request: pytest.FixtureRequest) -> None:
test = type(self).test_every_declared_field_becomes_a_property_in_field_order
request.applymarker(
meta(
Subject(
domain=Domain.SPEND_BUDGETS,
route=Route.CHAT_COMPLETIONS,
providers=(Provider.GEMINI, Provider.ANTHROPIC),
models=("gemini-2.5-flash", "claude-haiku-4-5"),
capabilities=(Capability.VISION, Capability.FUNCTION_CALLING, Capability.VISION),
mode=Mode.NONSTREAM,
)
)
)
assert subject_properties(collected_item(request, test.__name__)) == (
("domain", "spend-budgets"),
("route", "chat_completions"),
("provider", "anthropic"),
("provider", "gemini"),
("model", "claude-haiku-4-5"),
("model", "gemini-2.5-flash"),
("capability", "function_calling"),
("capability", "vision"),
("mode", "nonstream"),
)
def test_one_provider_with_three_models_pairs_nothing(self, request: pytest.FixtureRequest) -> None:
test = type(self).test_one_provider_with_three_models_pairs_nothing
request.applymarker(
meta(
Subject(
providers=(Provider.BEDROCK,),
models=("claude-sonnet-4-5", "claude-opus-4-7", "claude-haiku-4-5"),
)
)
)
assert subject_properties(collected_item(request, test.__name__)) == (
("provider", "bedrock"),
("model", "claude-haiku-4-5"),
("model", "claude-opus-4-7"),
("model", "claude-sonnet-4-5"),
)
def test_an_empty_plural_field_emits_nothing(self, request: pytest.FixtureRequest) -> None:
test = type(self).test_an_empty_plural_field_emits_nothing
request.applymarker(meta(Subject(domain=Domain.MANAGEMENT)))
assert subject_properties(collected_item(request, test.__name__)) == (("domain", "management"),)
def test_scalar_property_names_are_the_dataclass_field_names(self, request: pytest.FixtureRequest) -> None:
test = type(self).test_scalar_property_names_are_the_dataclass_field_names
request.applymarker(meta(Subject(domain=Domain.UNKNOWN, route=Route.HEALTH, mode=Mode.STREAM)))
declared = tuple(field.name for field in fields(Subject))
emitted = tuple(name for name, _ in subject_properties(collected_item(request, test.__name__)))
assert emitted == tuple(name for name in declared if name in {"domain", "route", "mode"})
def test_every_plural_field_is_deduped_and_sorted_at_declaration(self) -> None:
subject = Subject(
providers=(Provider.OPENAI, Provider.ANTHROPIC, Provider.OPENAI),
models=("gpt-5.5", "claude-haiku-4-5", "gpt-5.5"),
capabilities=(Capability.VISION, Capability.REASONING, Capability.VISION),
)
assert subject.providers == (Provider.ANTHROPIC, Provider.OPENAI)
assert subject.models == ("claude-haiku-4-5", "gpt-5.5")
assert subject.capabilities == (Capability.REASONING, Capability.VISION)
@pytest.mark.parametrize(
("field", "value"),
[
("models", "gpt-5.5"),
("models", ["gpt-5.5"]),
("providers", Provider.OPENAI),
("providers", [Provider.OPENAI]),
("capabilities", Capability.VISION),
("capabilities", frozenset({Capability.VISION})),
],
)
def test_a_plural_field_refuses_anything_but_a_tuple(self, field: str, value: object) -> None:
"""`replace` is the untyped way in, since the typed constructor would not let the test spell the mistake."""
with pytest.raises(TypeError, match=rf"Subject\.{field} must be a tuple"):
_ = replace(Subject(), **{field: value})
@pytest.mark.parametrize(
("field", "value", "member_type"),
[
("providers", ("openai",), "Provider"),
("capabilities", ("vision",), "Capability"),
("models", (5,), "str"),
],
)
def test_a_plural_field_refuses_a_member_of_the_wrong_type(
self, field: str, value: object, member_type: str
) -> None:
with pytest.raises(TypeError, match=rf"Subject\.{field} takes {member_type} members"):
_ = replace(Subject(), **{field: value})
def test_a_blank_model_is_dropped_rather_than_refused(self) -> None:
"""A blank env override must cost one missing property, not collection of the whole module."""
assert Subject(models=("", "gpt-5.5")).models == ("gpt-5.5",)
def test_the_typed_marker_only_ever_appends_to_the_fixed_prefix(self, request: pytest.FixtureRequest) -> None:
test = type(self).test_the_typed_marker_only_ever_appends_to_the_fixed_prefix
request.applymarker(pytest.mark.covers("quota_management.budget.key.blocks_over_limit"))
request.applymarker(meta(Subject(route=Route.SPEND_REPORTING)))
item = collected_item(request, test.__name__)
assert result_properties(item) == fixed_prefix(item, "quota_management.budget.key.blocks_over_limit") + (
("route", "spend_reporting"),
)
def test_a_test_with_only_the_old_string_covers_is_unchanged(self, request: pytest.FixtureRequest) -> None:
test = type(self).test_a_test_with_only_the_old_string_covers_is_unchanged
request.applymarker(pytest.mark.covers("llm.responses.openai.tool_use.nonstream.works"))
item = collected_item(request, test.__name__)
assert result_properties(item) == fixed_prefix(item, "llm.responses.openai.tool_use.nonstream.works")
def test_a_test_with_neither_marker_carries_only_the_prefix(self, request: pytest.FixtureRequest) -> None:
test = type(self).test_a_test_with_neither_marker_carries_only_the_prefix
item = collected_item(request, test.__name__)
assert subject_properties(item) == ()
assert result_properties(item) == fixed_prefix(item, "")
def test_a_marker_carrying_something_other_than_a_subject_emits_nothing(
self, request: pytest.FixtureRequest
) -> None:
test = type(self).test_a_marker_carrying_something_other_than_a_subject_emits_nothing
request.applymarker(pytest.mark.meta("spend-budgets"))
assert subject_properties(collected_item(request, test.__name__)) == ()
class TestProviderMirrorsLitellm:
"""`Provider` copies `LlmProviders` values so collecting tests/e2e never needs litellm; skips where it is absent."""
def test_every_provider_value_is_a_real_litellm_provider(self) -> None:
try:
from litellm.types.utils import LlmProviders
except ImportError: # pragma: no cover - the runner image's shape
pytest.skip("litellm is not importable here, which is the property under test")
known = {str(member.value) for member in LlmProviders}
unknown = sorted(member.value for member in Provider if member.value not in known)
assert not unknown, f"not LlmProviders values: {unknown}"
E2E_DIR: Final = Path(__file__).resolve().parents[1] / "e2e"
def _hand_typed_models(path: Path) -> Iterator[str]:
for node in ast.walk(ast.parse(path.read_text())):
match node:
case ast.Call(func=ast.Name(id="Subject"), keywords=keywords):
for keyword in keywords:
match keyword:
case ast.keyword(arg="models", value=ast.Tuple(elts=models)):
yield from (
f"{path.relative_to(E2E_DIR)}:{model.lineno} {model.value!r}"
for model in models
if isinstance(model, ast.Constant)
)
case _:
pass
case _:
pass
def test_a_declared_model_names_the_constant_the_test_drives() -> None:
offenders: Final = tuple(
offender for path in sorted(E2E_DIR.rglob("*.py")) for offender in _hand_typed_models(path)
)
assert offenders == ()
class TestStepRecording:
"""`@step`-decorated harness helpers append to the running test's story as
they execute.

View file

@ -131,6 +131,29 @@ Current limits: Bedrock cannot be mounted in record or replay (SigV4 signs the H
The harness is fully typed with no error budget: `make lint-e2e-basedpyright` must report zero basedpyright errors, and CI enforces that on any PR touching `tests/e2e/**/*.py`. When a response field is untyped, model it in `models.py` (just the fields you read) and let pydantic validate it, rather than threading a `dict` or `Any` through the test
## Typed test metadata
Separate from the coverage registry and additive to it: `@meta(Subject(...))` from `e2e_metadata.py` says what a test DRIVES, as closed enums rather than a string id. `@pytest.mark.covers("cell.id")` is untouched and keeps working exactly as before; the two markers coexist on the same test, and `@meta` always goes BELOW `@covers` so `Item.location` still anchors at the first decorator and every `source` deep link stays put
```python
@pytest.mark.covers("quota_management.budget.key.blocks_over_limit")
@meta(
Subject(
domain=Domain.SPEND_BUDGETS,
providers=(Provider.ANTHROPIC,),
models=(CHEAP_ANTHROPIC_MODEL,),
mode=Mode.NONSTREAM,
)
)
def test_bare_key_blocks_over_its_own_budget(...) -> None: ...
```
`route` is the endpoint the test is checking: `TEAM_MANAGEMENT` for a `/team/update` test, `SPEND_REPORTING` for a `/spend/logs` test, `MESSAGES` for a test of spend on `/v1/messages`. A test whose chat call only triggers the behavior under test, like the budget block above, leaves it unset, since its steps already name the call
Every field is optional today (the backfill of the rest of the suite is a later PR) and every field is a closed enum, so a typo is a basedpyright error at the call site rather than a property that silently never appears. `providers`, `models` and `capabilities` are tuples even with one member, because one test node routinely drives several: the claude_code matrix runs haiku, sonnet and opus in a single body, and a spend test calls two providers on one key. Declare every provider and every model the test drives, fallbacks included. The three are independent sets with no positional pairing between them (one provider x three models is the common case), and each is deduped and sorted at declaration so the committed run artifacts diff cleanly. `models=("gpt-5.5")` is a str and not a tuple, so anything but a tuple raises a `TypeError` where the decorator runs and shows up as a collection error naming the file. `Subject` is serialized with `dataclasses.asdict`, so a new scalar field needs no serializer edit; empty fields emit no `<property>` at all. A declared model names the constant the test drives (`CHEAP_ANTHROPIC_MODEL`, the file's own `BACKEND`), never a copy of its value, so the property cannot claim one model while an env override runs another. `e2e_metadata` and its call sites never import litellm, only the stdlib, pytest and pydantic: `Provider` mirrors litellm's `LlmProviders` values instead of importing them, because tests/e2e is shipped to the runner image on its own and a `from litellm...` at module scope would make the litellm package a hard dependency of COLLECTING the suite. `TestProviderMirrorsLitellm` in `tests/code_coverage_tests/test_e2e_metadata.py` fails on drift wherever litellm is importable and skips where it is not, so adding a provider is one line in `e2e_metadata`
Declared fields ride out as JUnit `<property>` entries behind the fixed prefix, the same way steps do: each scalar under its field name, and each plural value as a repeated property under its SINGULAR name (`provider`, `model`, `capability`). The results JSON downstream regroups them under the plural key, so `providers`, `models` and `capabilities` are arrays there, `[]` when empty
## Recorded test steps
`@step` from `e2e_metadata.py` goes on harness helpers (client methods and poll loops), never on a test. Each call adds one plain-English sentence to the running test's list of steps, in call order, so the list reads as what the test did. The step is recorded before the helper runs, so when a test fails, its last step is where it failed. Nobody writes steps by hand. They come from the calls the test actually made, so they can't drift from what happened

View file

@ -121,6 +121,11 @@ def pytest_configure(config: pytest.Config) -> None:
"markers",
"covers(cell_id, *, exercised_on=()): coverage-registry cell(s) this test covers",
)
config.addinivalue_line(
"markers",
"meta(subject): typed e2e_metadata.Subject describing what this test drives"
" (domain/route/providers/models/capabilities/mode); attach it with @meta(Subject(...))",
)
config.addinivalue_line(
"markers",
"replayable: edge-wired test whose provider traffic replays from a fixture bundle, so it makes "

View file

@ -1,14 +1,4 @@
"""Per-test metadata for the e2e suite: the step log each test records as it runs.
`steps` is appended at runtime by `@step`-decorated harness helpers, in call
order, so the list IS the test's user story and its last element is where a
failing test died. Nothing about it is hand-written, so it cannot drift from
what the test actually did.
tests/e2e is a black-box HTTP suite that imports litellm in zero files and is
shipped to the runner image as tests/e2e alone, and every harness module imports
this one, so it imports only the stdlib and pydantic.
"""
"""Typed per-test metadata for the e2e suite: what a test drives (`Subject`) and what it did (`steps`). See AGENTS.md"""
from __future__ import annotations
@ -20,13 +10,189 @@ import threading
from collections import deque
from collections.abc import Callable, Generator, Iterable, Mapping
from contextlib import AbstractContextManager, contextmanager
from dataclasses import asdict, dataclass
from enum import Enum
from functools import reduce, wraps
from types import TracebackType
from itertools import chain
from types import MappingProxyType, TracebackType
from typing import Final, ParamSpec, TypeVar, cast
import pytest
from pydantic import BaseModel
class Domain(str, Enum):
"""The OSS issue-label taxonomy, so an issue and a test join on one string"""
LLM_TRANSLATION = "llm-translation"
SPEND_BUDGETS = "spend-budgets"
UI = "ui"
MCP = "mcp"
OBSERVABILITY = "observability"
ROUTING = "routing"
DEPLOY_OPS = "deploy-ops"
COST_MAP = "cost-map"
PROXY_AUTH = "proxy-auth"
GUARDRAILS = "guardrails"
MANAGEMENT = "management"
SDK = "sdk"
PASSTHROUGH = "passthrough"
DB = "db"
CACHING = "caching"
DOCS = "docs"
AGENTS_API = "agents-api"
UNKNOWN = "unknown"
class Route(str, Enum):
"""The endpoint the test is checking; unset when the call only triggers the behavior under test"""
CHAT_COMPLETIONS = "chat_completions"
MESSAGES = "messages"
RESPONSES = "responses"
EMBEDDINGS = "embeddings"
COMPLETIONS = "completions"
FILES = "files"
BATCHES = "batches"
PASSTHROUGH = "passthrough"
MCP = "mcp"
GUARDRAILS = "guardrails"
KEY_MANAGEMENT = "key_management"
TEAM_MANAGEMENT = "team_management"
SPEND_REPORTING = "spend_reporting"
MODEL_MANAGEMENT = "model_management"
IMAGES = "images"
AUDIO = "audio"
MODERATIONS = "moderations"
RERANK = "rerank"
OCR = "ocr"
VECTOR_STORES = "vector_stores"
REALTIME = "realtime"
A2A = "a2a"
USER_MANAGEMENT = "user_management"
BUDGET_MANAGEMENT = "budget_management"
ORGANIZATION_MANAGEMENT = "organization_management"
CUSTOMER_MANAGEMENT = "customer_management"
HEALTH = "health"
METRICS = "metrics"
PROXY_CONFIG = "proxy_config"
ADMIN_UI = "admin_ui"
class Provider(str, Enum):
"""Mirrors litellm's `LlmProviders` without importing litellm; `TestProviderMirrorsLitellm` catches drift"""
OPENAI = "openai"
OPENAI_LIKE = "openai_like"
CUSTOM_OPENAI = "custom_openai"
AZURE = "azure"
AZURE_AI = "azure_ai"
ANTHROPIC = "anthropic"
GEMINI = "gemini"
VERTEX_AI = "vertex_ai"
BEDROCK = "bedrock"
SAGEMAKER = "sagemaker"
XAI = "xai"
GROQ = "groq"
DEEPSEEK = "deepseek"
MISTRAL = "mistral"
COHERE = "cohere"
PERPLEXITY = "perplexity"
OPENROUTER = "openrouter"
TOGETHER_AI = "together_ai"
FIREWORKS_AI = "fireworks_ai"
CEREBRAS = "cerebras"
SAMBANOVA = "sambanova"
NVIDIA_NIM = "nvidia_nim"
DATABRICKS = "databricks"
WATSONX = "watsonx"
OLLAMA = "ollama"
VLLM = "vllm"
HOSTED_VLLM = "hosted_vllm"
VOYAGE = "voyage"
JINA_AI = "jina_ai"
DEEPGRAM = "deepgram"
ELEVENLABS = "elevenlabs"
ASSEMBLYAI = "assemblyai"
LITELLM_PROXY = "litellm_proxy"
class Capability(str, Enum):
"""A model feature, 1:1 with a `supports_*` key in model_prices_and_context_window.json"""
FUNCTION_CALLING = "function_calling"
PARALLEL_FUNCTION_CALLING = "parallel_function_calling"
TOOL_CHOICE = "tool_choice"
TOOL_SEARCH = "tool_search"
VISION = "vision"
PDF_INPUT = "pdf_input"
AUDIO_INPUT = "audio_input"
REASONING = "reasoning"
WEB_SEARCH = "web_search"
PROMPT_CACHING = "prompt_caching"
RESPONSE_SCHEMA = "response_schema"
MID_CONVERSATION_SYSTEM = "mid_conversation_system"
class Mode(str, Enum):
"""How the route was driven"""
NONSTREAM = "nonstream"
STREAM = "stream"
BATCH = "batch"
WEBSOCKET = "websocket"
_M = TypeVar("_M")
def _scalar(value: object) -> str:
"""`str()` on a (str, Enum) gives `Route.RESPONSES`, and StrEnum needs 3.11"""
if isinstance(value, Enum):
return str(value.value) # pyright: ignore[reportAny] # Enum.value is Any for every enum
return str(value)
def _members(value: object) -> tuple[object, ...] | None:
return cast("tuple[object, ...]", value) if isinstance(value, tuple) else None
def _canonical(name: str, value: object, member_type: type[_M]) -> tuple[_M, ...]:
"""Validated, deduped and sorted; a bare str like `("gpt-5.5")` raises at import"""
members = _members(value)
if members is None:
raise TypeError(
f"Subject.{name} must be a tuple, got {type(value).__name__}: {value!r}."
f" A one-member tuple needs its trailing comma: {name}=(x,), not {name}=(x)"
)
typed = tuple(member for member in members if isinstance(member, member_type))
if len(typed) != len(members):
raise TypeError(f"Subject.{name} takes {member_type.__name__} members, got {value!r}")
return tuple(sorted(frozenset(member for member in typed if _scalar(member)), key=_scalar))
@dataclass(frozen=True, slots=True)
class Subject:
"""What a test is about. Not named `Test*` so pytest does not try to collect it"""
domain: Domain | None = None
route: Route | None = None
providers: tuple[Provider, ...] = ()
models: tuple[str, ...] = ()
capabilities: tuple[Capability, ...] = ()
mode: Mode | None = None
def __post_init__(self) -> None:
object.__setattr__(self, "providers", _canonical("providers", self.providers, Provider))
object.__setattr__(self, "models", _canonical("models", self.models, str))
object.__setattr__(self, "capabilities", _canonical("capabilities", self.capabilities, Capability))
def meta(subject: Subject) -> pytest.MarkDecorator:
"""Attach a `Subject` to a test: `@meta(Subject(route=Route.RESPONSES, ...))`"""
return pytest.mark.meta(subject)
_P = ParamSpec("_P")
_R = TypeVar("_R")
_Y = TypeVar("_Y")
@ -279,6 +445,35 @@ def step(label: str) -> Callable[[Callable[_P, _R]], Callable[_P, _R]]:
return decorate
_REPEATED: Final = MappingProxyType({"providers": "provider", "models": "model", "capabilities": "capability"})
def _declared_subject(args: tuple[object, ...]) -> Subject | None:
first = args[0] if args else None
return first if isinstance(first, Subject) else None
def subject_properties(item: pytest.Item) -> tuple[tuple[str, str], ...]:
"""The declared fields as <property> pairs, plural fields repeated under their singular name"""
marker: Final = item.get_closest_marker("meta")
if marker is None:
return ()
subject: Final = _declared_subject(marker.args)
if subject is None:
return ()
declared: Final[dict[str, object]] = asdict(subject)
return tuple(chain.from_iterable(_field_properties(name, value) for name, value in declared.items()))
def _field_properties(name: str, value: object) -> tuple[tuple[str, str], ...]:
repeated: Final = _REPEATED.get(name)
if repeated is not None:
return tuple((repeated, _scalar(member)) for member in _members(value) or ())
if value is None or value == "":
return ()
return ((name, _scalar(value)),)
def step_properties() -> tuple[tuple[str, str], ...]:
"""The step log as repeated `step` properties. Appended after the setup and
call phases, never at collection."""

View file

@ -20,7 +20,7 @@ from collections.abc import Iterable
import pytest
from coverage_registry.management_cases import case_properties
from e2e_metadata import step_properties
from e2e_metadata import step_properties, subject_properties
# Hardcoded because the runner image copies tests/e2e/ to /app/e2e, so nothing
# at runtime names this suite's place in the repo. test_junit_properties.py
@ -89,14 +89,16 @@ def covers_from_item(item: pytest.Item) -> tuple[str, ...]:
def result_properties(item: pytest.Item) -> tuple[tuple[str, str], ...]:
"""The custom signals a standard reporter cannot derive: the normalized suite
package, the comma-joined coverage-registry cell ids this test covers, and the
repo-relative `path:line` its source sits at."""
return (
"""The custom signals a standard reporter cannot derive.
Loki, Grafana and tests/integration/conftest.py read the `package`/`covers`/`source` prefix, so it never moves
"""
fixed = (
("package", package_from_nodeid(item.nodeid)),
("covers", ",".join(covers_from_item(item))),
("source", source_from_item(item)),
) + case_properties(item.nodeid)
)
return fixed + case_properties(item.nodeid) + subject_properties(item)
def attach_result_properties(item: pytest.Item) -> None:

View file

@ -5,6 +5,7 @@
addopts = --strict-markers --strict-config --reruns 1 --only-rerun "kind='network'" --only-rerun "status_code=5[0-9][0-9]"
markers =
e2e: live test that requires a running proxy and real provider keys
meta: typed e2e_metadata.Subject describing what this test drives (domain/route/providers/models/capabilities/mode); attach it with @meta(Subject(...)), never as a bare pytest.mark
replayable: edge-wired test whose provider traffic replays from a fixture bundle, so it makes zero provider calls in replay mode; the record/replay CI lane selects it with -m replayable
load: heavy throughput/load test; collected last so it never perturbs latency-sensitive suites
weekly: real-provider anomaly load test that spends real money; deselected unless E2E_WEEKLY_ANOMALY is set

View file

@ -10,12 +10,19 @@ from datetime import datetime, timezone
import pytest
from budget_client import BudgetClient
from e2e_metadata import Domain, Route, Subject, meta
from lifecycle import ResourceManager
pytestmark = pytest.mark.e2e
@pytest.mark.covers("mgmt.budget.new.persists")
@meta(
Subject(
domain=Domain.SPEND_BUDGETS,
route=Route.BUDGET_MANAGEMENT,
)
)
def test_budget_crud_roundtrip(client: BudgetClient, resources: ResourceManager) -> None:
budget_id = client.create_budget(max_budget=12.5, soft_budget=10.0, budget_duration="30d")
resources.defer(lambda: client.delete_budget(budget_id))
@ -38,6 +45,12 @@ def test_budget_crud_roundtrip(client: BudgetClient, resources: ResourceManager)
@pytest.mark.covers("mgmt.budget.delete.persists")
@meta(
Subject(
domain=Domain.SPEND_BUDGETS,
route=Route.BUDGET_MANAGEMENT,
)
)
def test_budget_delete_removes_it(client: BudgetClient, resources: ResourceManager) -> None:
budget_id = client.create_budget(max_budget=1.0)
resources.defer(lambda: client.delete_budget(budget_id))
@ -45,6 +58,12 @@ def test_budget_delete_removes_it(client: BudgetClient, resources: ResourceManag
assert not client.budget_info(budget_id), "budget still present after delete"
@meta(
Subject(
domain=Domain.SPEND_BUDGETS,
route=Route.KEY_MANAGEMENT,
)
)
def test_budget_duration_schedules_reset_on_key(client: BudgetClient, resources: ResourceManager) -> None:
key = client.generate_key(max_budget=10.0, budget_duration="30d")
resources.defer(lambda: client.delete_key(key))

View file

@ -19,16 +19,18 @@ import pytest
from budget_client import BudgetClient, is_budget_block
from e2e_config import unique_marker
from e2e_http import StreamingResponse, require_successful_call
from e2e_metadata import Domain, Mode, Provider, Subject, meta
from lifecycle import ResourceManager
pytestmark = pytest.mark.e2e
MODEL = "claude-haiku-4-5"
TINY_CAP = 3e-6
ROOMY_CAP = 100.0
def _chat(client: BudgetClient, key: str, *, user: str | None = None) -> StreamingResponse:
return client.chat(key, "claude-haiku-4-5", f"spend {unique_marker()}", max_tokens=16, user=user)
return client.chat(key, MODEL, f"spend {unique_marker()}", max_tokens=16, user=user)
def _assert_budget_blocks(client: BudgetClient, key: str, *, user: str = "") -> StreamingResponse:
@ -56,6 +58,14 @@ def _assert_blocked_422(client: BudgetClient, key: str) -> StreamingResponse:
class TestBudgetBlocksPerLevel:
@pytest.mark.covers("quota_management.budget.key.blocks_over_limit")
@meta(
Subject(
domain=Domain.SPEND_BUDGETS,
providers=(Provider.ANTHROPIC,),
models=(MODEL,),
mode=Mode.NONSTREAM,
)
)
def test_bare_key_blocks_over_its_own_budget(self, client: BudgetClient, resources: ResourceManager) -> None:
key = client.generate_key(max_budget=TINY_CAP)
resources.defer(lambda: client.delete_key(key))
@ -63,6 +73,14 @@ class TestBudgetBlocksPerLevel:
_assert_blocked_422(client, key)
@pytest.mark.covers("quota_management.budget.team.blocks_over_limit")
@meta(
Subject(
domain=Domain.SPEND_BUDGETS,
providers=(Provider.ANTHROPIC,),
models=(MODEL,),
mode=Mode.NONSTREAM,
)
)
def test_team_budget_blocks_every_team_key(self, client: BudgetClient, resources: ResourceManager) -> None:
team_id = client.create_team(alias=f"e2e-budget-team-{unique_marker()}", max_budget=TINY_CAP)
resources.defer(lambda: client.delete_team(team_id))
@ -79,6 +97,14 @@ class TestBudgetBlocksPerLevel:
)
@pytest.mark.covers("quota_management.budget.internal_user.blocks_over_limit")
@meta(
Subject(
domain=Domain.SPEND_BUDGETS,
providers=(Provider.ANTHROPIC,),
models=(MODEL,),
mode=Mode.NONSTREAM,
)
)
def test_user_budget_enforced_across_their_personal_keys(
self, client: BudgetClient, resources: ResourceManager
) -> None:
@ -113,18 +139,34 @@ class TestBudgetBlocksPerLevel:
require_successful_call(team_result)
@pytest.mark.covers("quota_management.budget.end_user.blocks_over_limit")
@meta(
Subject(
domain=Domain.SPEND_BUDGETS,
providers=(Provider.ANTHROPIC,),
models=(MODEL,),
mode=Mode.NONSTREAM,
)
)
def test_end_user_budget_blocks_attributed_calls(
self, client: BudgetClient, resources: ResourceManager
) -> None:
customer = f"e2e-budget-cust-{unique_marker()}"
client.create_customer(customer, max_budget=TINY_CAP)
resources.defer(lambda: client.delete_customers([customer]))
key = client.generate_key(models=["claude-haiku-4-5"])
key = client.generate_key(models=[MODEL])
resources.defer(lambda: client.delete_key(key))
_assert_budget_blocks(client, key, user=customer)
@pytest.mark.covers("quota_management.budget.organization.blocks_over_limit")
@meta(
Subject(
domain=Domain.SPEND_BUDGETS,
providers=(Provider.ANTHROPIC,),
models=(MODEL,),
mode=Mode.NONSTREAM,
)
)
def test_org_budget_blocks_keys_under_it(self, client: BudgetClient, resources: ResourceManager) -> None:
org_id = client.create_org(max_budget=TINY_CAP, alias=f"e2e-budget-org-{unique_marker()}")
resources.defer(lambda: client.delete_org(org_id))
@ -139,6 +181,14 @@ class TestBudgetBlocksPerLevel:
)
@pytest.mark.covers("quota_management.budget.team_member.blocks_over_limit")
@meta(
Subject(
domain=Domain.SPEND_BUDGETS,
providers=(Provider.ANTHROPIC,),
models=(MODEL,),
mode=Mode.NONSTREAM,
)
)
def test_member_budget_blocks_without_touching_teammates(
self, client: BudgetClient, resources: ResourceManager
) -> None:
@ -166,6 +216,14 @@ class TestKeyBudgetBlocksAcrossKeyKinds:
the capped key is refused, proving nothing around the key was the blocker."""
@pytest.mark.covers("quota_management.budget.key.blocks_over_limit")
@meta(
Subject(
domain=Domain.SPEND_BUDGETS,
providers=(Provider.ANTHROPIC,),
models=(MODEL,),
mode=Mode.NONSTREAM,
)
)
def test_personal_key_blocks_over_its_own_budget(
self, client: BudgetClient, resources: ResourceManager
) -> None:
@ -180,6 +238,14 @@ class TestKeyBudgetBlocksAcrossKeyKinds:
require_successful_call(_chat(client, control_key))
@pytest.mark.covers("quota_management.budget.key.blocks_over_limit")
@meta(
Subject(
domain=Domain.SPEND_BUDGETS,
providers=(Provider.ANTHROPIC,),
models=(MODEL,),
mode=Mode.NONSTREAM,
)
)
def test_team_key_blocks_over_its_own_budget(self, client: BudgetClient, resources: ResourceManager) -> None:
team_id = client.create_team(alias=f"e2e-key-cap-team-{unique_marker()}", max_budget=ROOMY_CAP)
resources.defer(lambda: client.delete_team(team_id))
@ -192,6 +258,14 @@ class TestKeyBudgetBlocksAcrossKeyKinds:
require_successful_call(_chat(client, control_key))
@pytest.mark.covers("quota_management.budget.key.blocks_over_limit")
@meta(
Subject(
domain=Domain.SPEND_BUDGETS,
providers=(Provider.ANTHROPIC,),
models=(MODEL,),
mode=Mode.NONSTREAM,
)
)
def test_team_member_key_blocks_over_its_own_budget(
self, client: BudgetClient, resources: ResourceManager
) -> None:

View file

@ -10,6 +10,7 @@ import pytest
from budget_client import BudgetClient, model_budget
from e2e_config import unique_marker
from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta
from lifecycle import ResourceManager
from models import AnthropicMessagesResponse
@ -20,6 +21,15 @@ FALLBACK_MODEL = "gpt-5.5"
@pytest.mark.covers("quota_management.budget.fallback.routes_to_fallback")
@meta(
Subject(
domain=Domain.SPEND_BUDGETS,
route=Route.MESSAGES,
providers=(Provider.ANTHROPIC, Provider.OPENAI),
models=(PRIMARY_MODEL, FALLBACK_MODEL),
mode=Mode.NONSTREAM,
)
)
def test_budget_fallback_reroutes_anthropic_messages_to_openai(
client: BudgetClient, resources: ResourceManager
) -> None:

View file

@ -22,11 +22,13 @@ import pytest
from budget_client import BudgetClient, is_budget_block
from e2e_config import unique_marker
from e2e_http import require_successful_call
from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta
from lifecycle import ResourceManager
from models import BudgetWindow
pytestmark = pytest.mark.e2e
MODEL = "claude-haiku-4-5"
WINDOW_SECONDS = 30
RESET_DEADLINE_SECONDS = 150
TINY_CAP = 3e-6
@ -34,7 +36,7 @@ SPEND_SETTLE_DEADLINE_SECONDS = 90
def _call(client: BudgetClient, key: str):
return client.chat(key, "claude-haiku-4-5", f"advance {unique_marker()}", max_tokens=16)
return client.chat(key, MODEL, f"advance {unique_marker()}", max_tokens=16)
def _poll_key_spend(client: BudgetClient, key: str, settled: Callable[[float], bool], problem: str) -> None:
@ -70,6 +72,12 @@ def _drive_to_block(client: BudgetClient, key: str) -> None:
# ---- Rung 1: scheduling exists at creation -----------------------------------
@meta(
Subject(
domain=Domain.SPEND_BUDGETS,
route=Route.KEY_MANAGEMENT,
)
)
def test_key_with_budget_duration_schedules_reset_at_creation(client: BudgetClient, resources: ResourceManager) -> None:
"""Baseline: a key created with a budget_duration has budget_reset_at populated
immediately. The reset job can only advance a timestamp that was scheduled in
@ -86,6 +94,14 @@ def test_key_with_budget_duration_schedules_reset_at_creation(client: BudgetClie
@pytest.mark.covers("quota_management.budget.key.blocks_over_limit")
@meta(
Subject(
domain=Domain.SPEND_BUDGETS,
providers=(Provider.ANTHROPIC,),
models=(MODEL,),
mode=Mode.NONSTREAM,
)
)
def test_key_spend_blocks_at_cap(client: BudgetClient, resources: ResourceManager) -> None:
"""Sanity that the tiny cap is enforced before we test that it resets: spend
accrues across calls and eventually returns budget_exceeded, never a 5xx."""
@ -103,6 +119,14 @@ def test_key_spend_blocks_at_cap(client: BudgetClient, resources: ResourceManage
@pytest.mark.covers("quota_management.budget.key.resets_after_window")
@meta(
Subject(
domain=Domain.SPEND_BUDGETS,
providers=(Provider.ANTHROPIC,),
models=(MODEL,),
mode=Mode.NONSTREAM,
)
)
def test_key_budget_reset_at_advances_after_window(client: BudgetClient, resources: ResourceManager) -> None:
"""The core #25109 guard: after the window elapses the reset job must move
budget_reset_at strictly forward AND zero key.spend. The broken nullable-JSON
@ -139,6 +163,14 @@ def test_key_budget_reset_at_advances_after_window(client: BudgetClient, resourc
@pytest.mark.covers("quota_management.budget.key_multi_window.resets_windows_independently")
@meta(
Subject(
domain=Domain.SPEND_BUDGETS,
providers=(Provider.ANTHROPIC,),
models=(MODEL,),
mode=Mode.NONSTREAM,
)
)
def test_multi_window_key_resets_each_window_independently(client: BudgetClient, resources: ResourceManager) -> None:
"""The JSON-backed path #25109 specifically touched. A tight 30s window and a
roomy 1m window: the tight window must reset on its own boundary while the roomy
@ -183,6 +215,14 @@ def test_multi_window_key_resets_each_window_independently(client: BudgetClient,
@pytest.mark.covers("quota_management.budget.team_member.resets_after_window")
@meta(
Subject(
domain=Domain.SPEND_BUDGETS,
providers=(Provider.ANTHROPIC,),
models=(MODEL,),
mode=Mode.NONSTREAM,
)
)
def test_team_member_budget_reset_at_advances(client: BudgetClient, resources: ResourceManager) -> None:
"""Per-team member windows are also JSON-backed. member_budget_reset_at must
advance after the window; the explicit before<after assertion is the #25109
@ -216,6 +256,14 @@ def test_team_member_budget_reset_at_advances(client: BudgetClient, resources: R
# ---- Rung 6: error-path edge - resets surface as blocks, never 5xx -----------
@meta(
Subject(
domain=Domain.SPEND_BUDGETS,
providers=(Provider.ANTHROPIC,),
models=(MODEL,),
mode=Mode.NONSTREAM,
)
)
def test_reset_wait_never_yields_non_budget_error(client: BudgetClient, resources: ResourceManager) -> None:
"""The other #25109 failure mode: a reset job that ERRORS on the nullable-JSON
column surfaces to the caller as a non-budget 5xx. Across the whole reset wait

View file

@ -7,10 +7,12 @@ import pytest
from budget_client import BudgetClient, is_budget_block
from e2e_config import unique_marker
from e2e_http import require_successful_call
from e2e_metadata import Domain, Mode, Provider, Subject, meta
from lifecycle import ResourceManager
pytestmark = pytest.mark.e2e
MODEL = "claude-haiku-4-5"
TINY_CAP = 3e-6
ROOMY_CAP = 100.0
WINDOW = "30s"
@ -18,7 +20,7 @@ RESET_DEADLINE_SECONDS = 150
def _call(client: BudgetClient, key: str):
return client.chat(key, "claude-haiku-4-5", f"reset {unique_marker()}", max_tokens=16)
return client.chat(key, MODEL, f"reset {unique_marker()}", max_tokens=16)
def _drive_to_block(client: BudgetClient, key: str) -> None:
@ -49,6 +51,14 @@ def _poll_until_serves_again(client: BudgetClient, key: str) -> None:
class TestBudgetResetPerLevel:
@pytest.mark.covers("quota_management.budget.key.resets_after_window")
@meta(
Subject(
domain=Domain.SPEND_BUDGETS,
providers=(Provider.ANTHROPIC,),
models=(MODEL,),
mode=Mode.NONSTREAM,
)
)
def test_bare_key_budget_resets_after_window(self, client: BudgetClient, resources: ResourceManager) -> None:
key = client.generate_key(max_budget=TINY_CAP, budget_duration=WINDOW)
resources.defer(lambda: client.delete_key(key))
@ -57,6 +67,14 @@ class TestBudgetResetPerLevel:
_poll_until_serves_again(client, key)
@pytest.mark.covers("quota_management.budget.team.resets_after_window")
@meta(
Subject(
domain=Domain.SPEND_BUDGETS,
providers=(Provider.ANTHROPIC,),
models=(MODEL,),
mode=Mode.NONSTREAM,
)
)
def test_team_budget_resets_after_window(self, client: BudgetClient, resources: ResourceManager) -> None:
team_id = client.create_team(
alias=f"e2e-team-reset-{unique_marker()}", max_budget=TINY_CAP, budget_duration=WINDOW
@ -69,6 +87,14 @@ class TestBudgetResetPerLevel:
_poll_until_serves_again(client, key)
@pytest.mark.covers("quota_management.budget.organization.resets_after_window")
@meta(
Subject(
domain=Domain.SPEND_BUDGETS,
providers=(Provider.ANTHROPIC,),
models=(MODEL,),
mode=Mode.NONSTREAM,
)
)
def test_org_budget_resets_after_window(self, client: BudgetClient, resources: ResourceManager) -> None:
org_id = client.create_org(
max_budget=TINY_CAP, alias=f"e2e-org-reset-{unique_marker()}", budget_duration=WINDOW
@ -91,6 +117,14 @@ class TestBudgetResetPerLevel:
_poll_until_serves_again(client, key)
@pytest.mark.covers("quota_management.budget.internal_user.resets_after_window")
@meta(
Subject(
domain=Domain.SPEND_BUDGETS,
providers=(Provider.ANTHROPIC,),
models=(MODEL,),
mode=Mode.NONSTREAM,
)
)
def test_personal_key_user_budget_resets_after_window(
self, client: BudgetClient, resources: ResourceManager
) -> None:
@ -109,6 +143,14 @@ class TestKeyBudgetResetAcrossKeyKinds:
the only thing that can block and the only thing that has to reset."""
@pytest.mark.covers("quota_management.budget.key.resets_after_window")
@meta(
Subject(
domain=Domain.SPEND_BUDGETS,
providers=(Provider.ANTHROPIC,),
models=(MODEL,),
mode=Mode.NONSTREAM,
)
)
def test_personal_key_resets_after_window(self, client: BudgetClient, resources: ResourceManager) -> None:
user_id = client.create_user(max_budget=ROOMY_CAP)
resources.defer(lambda: client.delete_user(user_id))
@ -119,6 +161,14 @@ class TestKeyBudgetResetAcrossKeyKinds:
_poll_until_serves_again(client, key)
@pytest.mark.covers("quota_management.budget.key.resets_after_window")
@meta(
Subject(
domain=Domain.SPEND_BUDGETS,
providers=(Provider.ANTHROPIC,),
models=(MODEL,),
mode=Mode.NONSTREAM,
)
)
def test_team_key_resets_after_window(self, client: BudgetClient, resources: ResourceManager) -> None:
team_id = client.create_team(alias=f"e2e-key-reset-team-{unique_marker()}", max_budget=ROOMY_CAP)
resources.defer(lambda: client.delete_team(team_id))
@ -129,6 +179,14 @@ class TestKeyBudgetResetAcrossKeyKinds:
_poll_until_serves_again(client, key)
@pytest.mark.covers("quota_management.budget.key.resets_after_window")
@meta(
Subject(
domain=Domain.SPEND_BUDGETS,
providers=(Provider.ANTHROPIC,),
models=(MODEL,),
mode=Mode.NONSTREAM,
)
)
def test_team_member_key_resets_after_window(self, client: BudgetClient, resources: ResourceManager) -> None:
team_id = client.create_team(alias=f"e2e-key-reset-team-{unique_marker()}", max_budget=ROOMY_CAP)
resources.defer(lambda: client.delete_team(team_id))

View file

@ -21,6 +21,7 @@ import pytest
from budget_client import BudgetClient, is_budget_block
from e2e_config import unique_marker
from e2e_http import StreamingResponse, require_successful_call
from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta
from lifecycle import ResourceManager
from models import KeyGenerateBody, LiteLLMParamsBody, ModelInfoBody, ModelNewBody
@ -102,6 +103,14 @@ def drained(client: BudgetClient) -> Iterator[DrainedPool]:
class TestModelAccessGroupBudget:
@pytest.mark.covers("quota_management.budget.model_access_group.blocks_over_limit")
@meta(
Subject(
domain=Domain.SPEND_BUDGETS,
providers=(Provider.OPENAI,),
models=(BACKEND,),
mode=Mode.NONSTREAM,
)
)
def test_the_key_that_drained_the_pool_stays_blocked(
self, client: BudgetClient, drained: DrainedPool
) -> None:
@ -114,6 +123,14 @@ class TestModelAccessGroupBudget:
)
@pytest.mark.covers("quota_management.budget.model_access_group.enforced_across_keys")
@meta(
Subject(
domain=Domain.SPEND_BUDGETS,
providers=(Provider.OPENAI,),
models=(BACKEND,),
mode=Mode.NONSTREAM,
)
)
def test_a_key_that_spent_nothing_is_blocked_by_the_shared_pool(
self, client: BudgetClient, resources: ResourceManager, drained: DrainedPool
) -> None:
@ -127,6 +144,14 @@ class TestModelAccessGroupBudget:
)
@pytest.mark.covers("quota_management.budget.model_access_group.isolates_per_group")
@meta(
Subject(
domain=Domain.SPEND_BUDGETS,
providers=(Provider.OPENAI,),
models=(BACKEND,),
mode=Mode.NONSTREAM,
)
)
def test_a_drained_group_does_not_block_a_different_group(
self, client: BudgetClient, resources: ResourceManager, drained: DrainedPool
) -> None:
@ -141,6 +166,14 @@ class TestModelAccessGroupBudget:
require_successful_call(result)
@pytest.mark.covers("quota_management.budget.model_access_group.reports_spend")
@meta(
Subject(
domain=Domain.SPEND_BUDGETS,
route=Route.BUDGET_MANAGEMENT,
providers=(Provider.OPENAI,),
models=(BACKEND,),
)
)
def test_the_budget_read_reports_the_spend_drawn_against_the_pool(
self, client: BudgetClient, drained: DrainedPool
) -> None:

View file

@ -13,6 +13,7 @@ import pytest
from budget_client import BudgetClient, is_budget_block, model_budget
from e2e_config import unique_marker
from e2e_http import require_successful_call
from e2e_metadata import Domain, Mode, Provider, Subject, meta
from lifecycle import ResourceManager
from models import ModelBudgetEntry
@ -30,6 +31,14 @@ def _call(client: BudgetClient, key: str, model: str):
@pytest.mark.covers("quota_management.budget.model_max.isolates_per_model")
@meta(
Subject(
domain=Domain.SPEND_BUDGETS,
providers=(Provider.ANTHROPIC, Provider.GEMINI),
models=(CAPPED_MODEL, FREE_MODEL),
mode=Mode.NONSTREAM,
)
)
def test_model_max_budget_isolates_per_model(
client: BudgetClient, resources: ResourceManager
) -> None:
@ -61,6 +70,14 @@ def test_model_max_budget_isolates_per_model(
@pytest.mark.skip(reason="stage red: product gap, end-user model_max_budget rpm_limit is stored but never enforced")
@pytest.mark.covers("quota_management.budget.end_user_model_max.blocks_over_limit")
@meta(
Subject(
domain=Domain.SPEND_BUDGETS,
providers=(Provider.GEMINI,),
models=(FREE_MODEL,),
mode=Mode.NONSTREAM,
)
)
def test_end_user_model_max_budget_enforces_per_model_rpm(
client: BudgetClient, resources: ResourceManager
) -> None:

View file

@ -22,6 +22,7 @@ import pytest
from budget_client import BudgetClient, is_budget_block, window_reset_at
from e2e_http import StreamingResponse, require_successful_call
from e2e_config import CHEAP_OPENAI_MODEL, unique_marker
from e2e_metadata import Domain, Mode, Provider, Subject, meta
from lifecycle import ResourceManager
from models import BudgetWindow
@ -57,6 +58,14 @@ def _drive_to_block(client: BudgetClient, key: str) -> StreamingResponse:
@pytest.mark.covers("quota_management.budget.key_multi_window.blocks_then_resets")
@meta(
Subject(
domain=Domain.SPEND_BUDGETS,
providers=(Provider.OPENAI,),
models=(CHEAP_OPENAI_MODEL,),
mode=Mode.NONSTREAM,
)
)
def test_short_window_blocks_then_resets(client: BudgetClient, resources: ResourceManager) -> None:
key = client.generate_key(
models=[MODEL],
@ -90,6 +99,14 @@ def test_short_window_blocks_then_resets(client: BudgetClient, resources: Resour
@pytest.mark.covers("quota_management.budget.key_multi_window.blocks_then_resets")
@meta(
Subject(
domain=Domain.SPEND_BUDGETS,
providers=(Provider.OPENAI,),
models=(CHEAP_OPENAI_MODEL,),
mode=Mode.NONSTREAM,
)
)
def test_long_window_blocks_after_short_window_resets(client: BudgetClient, resources: ResourceManager) -> None:
key = client.generate_key(
models=[MODEL],

View file

@ -12,12 +12,23 @@ import pytest
from budget_client import BudgetClient, is_budget_block
from e2e_config import unique_marker
from e2e_http import require_successful_call
from e2e_metadata import Domain, Mode, Provider, Subject, meta
from lifecycle import ResourceManager
pytestmark = pytest.mark.e2e
MODEL = "claude-haiku-4-5"
@pytest.mark.covers("quota_management.budget.soft.alerts_without_blocking")
@meta(
Subject(
domain=Domain.SPEND_BUDGETS,
providers=(Provider.ANTHROPIC,),
models=(MODEL,),
mode=Mode.NONSTREAM,
)
)
def test_soft_budget_does_not_block(
client: BudgetClient, resources: ResourceManager
) -> None:
@ -27,7 +38,7 @@ def test_soft_budget_does_not_block(
for _ in range(3):
result = client.chat(
key, "claude-haiku-4-5", f"hi {unique_marker()}", max_tokens=16
key, MODEL, f"hi {unique_marker()}", max_tokens=16
)
assert not is_budget_block(result), (
"soft_budget blocked a request; it must alert only, not block "

View file

@ -28,6 +28,7 @@ from pydantic import TypeAdapter, ValidationError
from budget_client import BudgetClient
from e2e_config import unique_marker
from e2e_http import StreamingResponse
from e2e_metadata import Domain, Mode, Provider, Subject, meta
from lifecycle import ResourceManager
if TYPE_CHECKING:
@ -144,6 +145,14 @@ def _accumulate(client: BudgetClient, key: str, count: int) -> None:
@pytest.mark.covers("quota_management.budget.spend_counter.reseed_matches_db")
@meta(
Subject(
domain=Domain.SPEND_BUDGETS,
providers=(Provider.ANTHROPIC,),
models=(MODEL,),
mode=Mode.NONSTREAM,
)
)
def test_cold_counter_reseed_keeps_counter_equal_to_db_spend(
client: BudgetClient, resources: ResourceManager
) -> None:

View file

@ -13,17 +13,19 @@ import pytest
from budget_client import BudgetClient, is_budget_block
from e2e_config import unique_marker
from e2e_http import require_successful_call
from e2e_metadata import Domain, Mode, Provider, Subject, meta
from lifecycle import ResourceManager
pytestmark = pytest.mark.e2e
MODEL = "claude-haiku-4-5"
TINY_BUDGET = 1e-6
def _tagged_call(client: BudgetClient, key: str, tag: str):
result = client.chat(
key,
"claude-haiku-4-5",
MODEL,
f"hi {unique_marker()}",
tags=[tag],
max_tokens=64,
@ -34,6 +36,14 @@ def _tagged_call(client: BudgetClient, key: str, tag: str):
@pytest.mark.covers("quota_management.budget.tag.blocks_over_limit")
@meta(
Subject(
domain=Domain.SPEND_BUDGETS,
providers=(Provider.ANTHROPIC,),
models=(MODEL,),
mode=Mode.NONSTREAM,
)
)
def test_tag_budget_blocks_tagged_requests(
client: BudgetClient, scoped_key: str, resources: ResourceManager
) -> None:

View file

@ -21,6 +21,7 @@ import pytest
from budget_client import BudgetClient, is_budget_block
from e2e_config import unique_marker
from e2e_http import Success, require_successful_call
from e2e_metadata import Domain, Mode, Provider, Subject, meta
from lifecycle import ResourceManager
from models import ChatBody, ChatMessage
@ -79,6 +80,14 @@ def _send(client: BudgetClient, key: str) -> str | None:
class TestTeamMemberBudget:
@meta(
Subject(
domain=Domain.SPEND_BUDGETS,
providers=(Provider.ANTHROPIC,),
models=(MODEL,),
mode=Mode.NONSTREAM,
)
)
def test_member_spend_attributed_to_team_and_user(self, client: BudgetClient, member: _Member) -> None:
sent = frozenset(rid for rid in (_send(client, member.key) for _ in range(BURST)) if rid)
assert sent, "no member call went through; cannot check attribution"
@ -98,6 +107,14 @@ class TestTeamMemberBudget:
)
@pytest.mark.covers("quota_management.budget.team_member.blocks_over_limit")
@meta(
Subject(
domain=Domain.SPEND_BUDGETS,
providers=(Provider.ANTHROPIC,),
models=(MODEL,),
mode=Mode.NONSTREAM,
)
)
def test_member_spend_over_budget_is_blocked(self, client: BudgetClient, member: _Member) -> None:
for _ in range(40):
result = client.chat(member.key, MODEL, f"spend {unique_marker()}", max_tokens=16)

View file

@ -17,6 +17,7 @@ import pytest
from budget_client import BudgetClient, is_budget_block
from e2e_config import unique_marker
from e2e_http import Success, require_successful_call
from e2e_metadata import Domain, Mode, Provider, Subject, meta
from lifecycle import ResourceManager
from models import ChatBody, ChatMessage
@ -88,6 +89,14 @@ def _roomy_send(client: BudgetClient, key: str) -> str:
class TestTeamMemberBudgetIsolation:
@pytest.mark.covers("quota_management.budget.team_member.isolates_per_member")
@meta(
Subject(
domain=Domain.SPEND_BUDGETS,
providers=(Provider.OPENAI,),
models=(MODEL,),
mode=Mode.NONSTREAM,
)
)
def test_blocked_member_does_not_block_peer(self, client: BudgetClient, pair: _Pair) -> None:
blocked = False
for _ in range(40):

View file

@ -6,10 +6,12 @@ import pytest
from budget_client import BudgetClient
from e2e_config import unique_marker
from e2e_http import require_successful_call
from e2e_metadata import Domain, Mode, Provider, Subject, meta
from lifecycle import ResourceManager
pytestmark = pytest.mark.e2e
MODEL = "claude-haiku-4-5"
MEMBER_BUDGET = 1.0 # default member budget is $50, we're testing with a smaller value
def _as_datetime(value: str) -> datetime:
@ -17,6 +19,14 @@ def _as_datetime(value: str) -> datetime:
@pytest.mark.covers("quota_management.budget.team_member.resets_after_window")
@meta(
Subject(
domain=Domain.SPEND_BUDGETS,
providers=(Provider.ANTHROPIC,),
models=(MODEL,),
mode=Mode.NONSTREAM,
)
)
def test_team_member_budget_reset_keeps_advancing(client: BudgetClient, resources: ResourceManager) -> None:
team_id = client.create_team(alias=f"e2e-member-reset-{unique_marker()}", max_budget=100.0)
resources.defer(lambda: client.delete_team(team_id))
@ -34,7 +44,7 @@ def test_team_member_budget_reset_keeps_advancing(client: BudgetClient, resource
# the member can spend within the team while the window is live
key = client.generate_key(team_id=team_id, user_id=user_id)
resources.defer(lambda: client.delete_key(key))
require_successful_call(client.chat(key, "claude-haiku-4-5", f"reset {unique_marker()}", max_tokens=16))
require_successful_call(client.chat(key, MODEL, f"reset {unique_marker()}", max_tokens=16))
# once the window elapses the reset job must move budget_reset_at forward; a job
# that skips the member's budget row (the #25109 regression) leaves it pinned at

View file

@ -24,11 +24,13 @@ import pytest
from budget_client import BudgetClient, is_budget_block, window_reset_at
from e2e_http import StreamingResponse, require_successful_call
from e2e_config import unique_marker
from e2e_metadata import Domain, Mode, Provider, Subject, meta
from lifecycle import ResourceManager
from models import BudgetWindow
pytestmark = pytest.mark.e2e
MODEL = "claude-haiku-4-5"
WINDOW_SECONDS = 30
SHORT_WINDOW = f"{WINDOW_SECONDS}s"
LONG_WINDOW = "1d"
@ -38,7 +40,7 @@ RESET_DEADLINE_SECONDS = 150
def _call(client: BudgetClient, key: str):
return client.chat(key, "claude-haiku-4-5", f"team-window {unique_marker()}", max_tokens=16)
return client.chat(key, MODEL, f"team-window {unique_marker()}", max_tokens=16)
def _drive_to_block(client: BudgetClient, key: str) -> StreamingResponse:
@ -52,6 +54,14 @@ def _drive_to_block(client: BudgetClient, key: str) -> StreamingResponse:
@pytest.mark.covers("quota_management.budget.team_multi_window.blocks_then_resets")
@meta(
Subject(
domain=Domain.SPEND_BUDGETS,
providers=(Provider.ANTHROPIC,),
models=(MODEL,),
mode=Mode.NONSTREAM,
)
)
def test_team_short_window_blocks_then_resets(client: BudgetClient, resources: ResourceManager) -> None:
team_id = client.create_team(
alias=f"e2e-team-window-{unique_marker()}",
@ -61,7 +71,7 @@ def test_team_short_window_blocks_then_resets(client: BudgetClient, resources: R
],
)
resources.defer(lambda: client.delete_team(team_id))
key = client.generate_key(team_id=team_id, models=["claude-haiku-4-5"])
key = client.generate_key(team_id=team_id, models=[MODEL])
resources.defer(lambda: client.delete_key(key))
# 1. exhaust the tight window -> litellm returns budget_exceeded
@ -85,6 +95,14 @@ def test_team_short_window_blocks_then_resets(client: BudgetClient, resources: R
@pytest.mark.covers("quota_management.budget.team_multi_window.blocks_then_resets")
@meta(
Subject(
domain=Domain.SPEND_BUDGETS,
providers=(Provider.ANTHROPIC,),
models=(MODEL,),
mode=Mode.NONSTREAM,
)
)
def test_team_long_window_blocks_after_short_window_resets(client: BudgetClient, resources: ResourceManager) -> None:
# 0. key with a short budget window and a long budget window
@ -96,7 +114,7 @@ def test_team_long_window_blocks_after_short_window_resets(client: BudgetClient,
],
)
resources.defer(lambda: client.delete_team(team_id))
key = client.generate_key(team_id=team_id, models=["claude-haiku-4-5"])
key = client.generate_key(team_id=team_id, models=[MODEL])
resources.defer(lambda: client.delete_key(key))
# 1. drive the key to being blocked, assert its blocked by budget budget_exceeded

View file

@ -15,6 +15,7 @@ import pytest
from budget_client import BudgetClient, is_budget_block
from e2e_config import unique_marker
from e2e_http import StreamingResponse, require_successful_call
from e2e_metadata import Domain, Mode, Provider, Subject, meta
from lifecycle import ResourceManager
pytestmark = pytest.mark.e2e
@ -58,6 +59,14 @@ def _expect_prompt_block(client: BudgetClient, key: str, subject: str) -> None:
class TestUserBudgetAcrossKeys:
@pytest.mark.covers("quota_management.budget.internal_user.enforced_across_keys")
@meta(
Subject(
domain=Domain.SPEND_BUDGETS,
providers=(Provider.OPENAI,),
models=(MODEL,),
mode=Mode.NONSTREAM,
)
)
def test_user_budget_blocks_a_second_key(self, client: BudgetClient, resources: ResourceManager) -> None:
user_id = client.create_user(max_budget=TINY_CAP)
resources.defer(lambda: client.delete_user(user_id))

View file

@ -46,6 +46,7 @@ from pydantic import BaseModel, ConfigDict, ValidationError
from e2e_config import unique_marker
from e2e_http import StreamingResponse, require_successful_call
from e2e_metadata import Domain, Mode, Provider, Subject, meta
from lifecycle import ResourceManager
from models import KeyGenerateBody, KeyMetadata, LiteLLMParamsBody
from quota_client import QuotaClient
@ -157,6 +158,14 @@ class TestDynamicRateLimitPriority:
"quota_management.ratelimit.priority_generous.picks_under_tpm",
exercised_on=["chat_completions"],
)
@meta(
Subject(
domain=Domain.SPEND_BUDGETS,
providers=(Provider.ANTHROPIC,),
models=(BACKEND,),
mode=Mode.NONSTREAM,
)
)
def test_generous_mode_lets_priority_borrow_past_reservation(
self, client: QuotaClient, resources: ResourceManager
) -> None:
@ -199,6 +208,14 @@ class TestDynamicRateLimitPriority:
"quota_management.ratelimit.priority_strict.picks_under_tpm",
exercised_on=["chat_completions"],
)
@meta(
Subject(
domain=Domain.SPEND_BUDGETS,
providers=(Provider.ANTHROPIC,),
models=(BACKEND,),
mode=Mode.NONSTREAM,
)
)
def test_strict_mode_blocks_saturated_priority_but_serves_the_other(
self, client: QuotaClient, resources: ResourceManager
) -> None:

View file

@ -39,6 +39,7 @@ from pydantic import BaseModel, ConfigDict, ValidationError
from e2e_config import CHEAP_ANTHROPIC_MODEL, unique_marker
from e2e_http import StreamingResponse, require_successful_call
from e2e_metadata import Domain, Mode, Provider, Subject, meta
from lifecycle import ResourceManager
from models import KeyGenerateBody
from quota_client import QuotaClient
@ -176,6 +177,14 @@ def _assert_rate_limited(outcome: StreamingResponse, limit_type: str) -> None:
class TestKeyRateLimits:
@pytest.mark.covers("quota_management.ratelimit.rpm.blocks_over_limit")
@meta(
Subject(
domain=Domain.SPEND_BUDGETS,
providers=(Provider.ANTHROPIC,),
models=(CHEAP_ANTHROPIC_MODEL,),
mode=Mode.NONSTREAM,
)
)
def test_rpm_limit_blocks_over_limit(self, client: QuotaClient, resources: ResourceManager) -> None:
key = _limited_key(client, resources, rpm_limit=3)
info = client.proxy.key_info(key)
@ -188,6 +197,14 @@ class TestKeyRateLimits:
_assert_rate_limited(_chat(client, key), "requests")
@pytest.mark.covers("quota_management.ratelimit.tpm.blocks_over_limit")
@meta(
Subject(
domain=Domain.SPEND_BUDGETS,
providers=(Provider.ANTHROPIC,),
models=(CHEAP_ANTHROPIC_MODEL,),
mode=Mode.NONSTREAM,
)
)
def test_tpm_limit_blocks_over_limit(self, client: QuotaClient, resources: ResourceManager) -> None:
key = _limited_key(client, resources, tpm_limit=TPM_LIMIT)
info = client.proxy.key_info(key)
@ -207,6 +224,14 @@ class TestKeyRateLimits:
_assert_rate_limited(_chat(client, key), "tokens")
@pytest.mark.covers("quota_management.ratelimit.rpm.resets_after_window")
@meta(
Subject(
domain=Domain.SPEND_BUDGETS,
providers=(Provider.ANTHROPIC,),
models=(CHEAP_ANTHROPIC_MODEL,),
mode=Mode.NONSTREAM,
)
)
def test_rpm_limit_resets_after_window(self, client: QuotaClient, resources: ResourceManager) -> None:
key = _limited_key(client, resources, rpm_limit=1)
@ -232,6 +257,14 @@ class TestKeyRateLimits:
pytest.fail("a blocked key never recovered after the rate-limit window elapsed")
@pytest.mark.covers("quota_management.ratelimit.rpm.headers_report_remaining")
@meta(
Subject(
domain=Domain.SPEND_BUDGETS,
providers=(Provider.ANTHROPIC,),
models=(CHEAP_ANTHROPIC_MODEL,),
mode=Mode.NONSTREAM,
)
)
def test_headers_report_limit_and_remaining(self, client: QuotaClient, resources: ResourceManager) -> None:
key = _limited_key(client, resources, rpm_limit=5, tpm_limit=100000)

View file

@ -13,6 +13,7 @@ import pytest
from e2e_config import unique_marker
from e2e_http import require_successful_call
from e2e_metadata import Domain, Mode, Provider, Subject, meta
from lifecycle import ResourceManager
from models import KeyGenerateBody, LiteLLMParamsBody
from quota_client import QuotaClient
@ -40,6 +41,14 @@ class TestRedisBackedRateLimit:
"quota_management.ratelimit.redis_backed.blocks_over_limit",
exercised_on=["chat_completions"],
)
@meta(
Subject(
domain=Domain.SPEND_BUDGETS,
providers=(Provider.ANTHROPIC,),
models=(BACKEND,),
mode=Mode.NONSTREAM,
)
)
def test_rpm_limit_one_blocks_second_call(
self, client: QuotaClient, resources: ResourceManager
) -> None:

View file

@ -15,6 +15,7 @@ import pytest
from e2e_config import unique_marker
from e2e_http import require_successful_call
from e2e_metadata import Domain, Mode, Provider, Subject, meta
from lifecycle import ResourceManager
from models import KeyGenerateBody, LiteLLMParamsBody
from quota_client import QuotaClient
@ -45,6 +46,14 @@ class TestRedisCircuitBreakerPath:
"reliability.circuit_breaker.redis.trips_then_recovers",
exercised_on=["chat_completions"],
)
@meta(
Subject(
domain=Domain.SPEND_BUDGETS,
providers=(Provider.ANTHROPIC,),
models=(BACKEND,),
mode=Mode.NONSTREAM,
)
)
def test_burst_rate_limit_does_not_freeze_fresh_key(
self, client: QuotaClient, resources: ResourceManager
) -> None:

View file

@ -24,6 +24,7 @@ from models import (
TextBlock,
Usage,
)
from e2e_metadata import Capability, Domain, Mode, Provider, Subject, meta
from quota_client import QuotaClient
pytestmark = [pytest.mark.e2e, pytest.mark.provider_live]
@ -101,6 +102,15 @@ class TestTpmExcludesCachedTokens:
"quota_management.ratelimit.tpm.excludes_cached_tokens",
exercised_on=["chat_completions"],
)
@meta(
Subject(
domain=Domain.SPEND_BUDGETS,
providers=(Provider.ANTHROPIC,),
models=(ANTHROPIC_MODEL,),
capabilities=(Capability.PROMPT_CACHING,),
mode=Mode.NONSTREAM,
)
)
def test_cache_hit_reduces_tpm_by_non_cached_only(
self, client: QuotaClient, resources: ResourceManager
) -> None:

View file

@ -9,6 +9,7 @@ from lifecycle import ResourceManager
from models import ChatBody, ChatMessage, ChatResponse, KeyGenerateBody, LiteLLMParamsBody, TeamNewBody
from spend_e2e_client import SpendClient
BACKEND: Final = "openai/gpt-5.6-luna"
INPUT_RATE: Final = 0.00004
OUTPUT_RATE: Final = 0.00008
@ -38,7 +39,7 @@ def create_traffic(client: SpendClient, resources: ResourceManager) -> tuple[Tea
model_id: Final = client.proxy.create_model(
model,
LiteLLMParamsBody(
model="openai/gpt-5.6-luna",
model=BACKEND,
api_key="os.environ/OPENAI_API_KEY",
api_base=None if base is None else f"{base}/v1",
input_cost_per_token=INPUT_RATE,

View file

@ -52,6 +52,7 @@ from cost_rows import (
)
from e2e_config import unique_marker
from e2e_http import unwrap
from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta
from lifecycle import ResourceManager
from models import AnthropicMessagesBody, ChatBody, ChatMessage, LiteLLMParamsBody
from pydantic import BaseModel
@ -122,6 +123,15 @@ def _assert_cache_read_billed(row: CostRow) -> None:
class TestCacheCostAccounting:
@pytest.mark.covers("quota_management.spend_tracking.cache_write.bills_cache_creation_rate")
@meta(
Subject(
domain=Domain.SPEND_BUDGETS,
providers=(Provider.OPENAI,),
models=(CACHE_WRITE_BACKEND,),
capabilities=(Capability.PROMPT_CACHING,),
mode=Mode.NONSTREAM,
)
)
def test_cache_write_tokens_billed_at_cache_creation_rate(
self, client: SpendClient, resources: ResourceManager, scoped_key: str
) -> None:
@ -152,6 +162,15 @@ class TestCacheCostAccounting:
assert_total_is_sum_of_components(row)
@pytest.mark.covers("quota_management.spend_tracking.cost_breakdown.reports_component_costs")
@meta(
Subject(
domain=Domain.SPEND_BUDGETS,
providers=(Provider.OPENAI,),
models=(CACHE_READ_BACKEND,),
capabilities=(Capability.PROMPT_CACHING, Capability.REASONING),
mode=Mode.NONSTREAM,
)
)
def test_cost_breakdown_reports_component_costs(
self, client: SpendClient, resources: ResourceManager, scoped_key: str
) -> None:
@ -216,6 +235,15 @@ class TestCacheCostAccounting:
_assert_cache_read_billed(row)
@pytest.mark.covers("quota_management.spend_tracking.stream_cache_read.bills_cache_read_rate")
@meta(
Subject(
domain=Domain.SPEND_BUDGETS,
providers=(Provider.OPENAI,),
models=(CACHE_READ_BACKEND,),
capabilities=(Capability.PROMPT_CACHING,),
mode=Mode.STREAM,
)
)
def test_streaming_cache_read_billed_at_cache_read_rate(
self, client: SpendClient, resources: ResourceManager, scoped_key: str
) -> None:
@ -247,6 +275,16 @@ class TestCacheCostAccounting:
_assert_cache_read_billed(row)
@pytest.mark.covers("quota_management.spend_tracking.messages_bridge.keeps_cache_tokens")
@meta(
Subject(
domain=Domain.SPEND_BUDGETS,
route=Route.MESSAGES,
providers=(Provider.OPENAI,),
models=(BRIDGE_BACKEND,),
capabilities=(Capability.PROMPT_CACHING,),
mode=Mode.NONSTREAM,
)
)
def test_messages_bridge_keeps_cache_tokens(
self, client: SpendClient, resources: ResourceManager, scoped_key: str
) -> None:

View file

@ -27,6 +27,7 @@ import pytest
from cost_rows import approx_equal, cacheable_prefix, register_priced_model
from e2e_config import unique_marker
from e2e_http import StreamingResponse
from e2e_metadata import Domain, Mode, Provider, Subject, meta
from lifecycle import ResourceManager
from models import ChatBody, ChatMessage, ChatResponse, LiteLLMParamsBody
from spend_e2e_client import SpendClient
@ -60,6 +61,14 @@ def _header_cost(response: StreamingResponse, name: str) -> float:
class TestCostHeaders:
@pytest.mark.covers("quota_management.spend_tracking.cost_headers.additive_components")
@meta(
Subject(
domain=Domain.SPEND_BUDGETS,
providers=(Provider.OPENAI,),
models=(BACKEND,),
mode=Mode.NONSTREAM,
)
)
def test_component_cost_headers_sum_to_total(
self, client: SpendClient, resources: ResourceManager, scoped_key: str
) -> None:

View file

@ -36,6 +36,7 @@ from datetime import datetime, timedelta, timezone
from typing import Final
import pytest
from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta
from models import KeyGenerateBody
from proxy_client import Converged, await_converged
from pydantic import BaseModel
@ -61,6 +62,7 @@ EMBED_MODEL: Final = "openai-text-embedding-3-small"
BATCH_MODEL: Final = "openai-gpt-4o-mini"
BATCH_BACKEND_MODEL: Final = "gpt-4o-mini"
BATCH_PROVIDER: Final = "openai"
DRIVEN_MODELS: Final = (CHAT_MODEL, MESSAGES_MODEL, RESPONSES_MODEL, EMBED_MODEL, BATCH_MODEL)
HEALTH_SERVICE_ACCOUNT: Final = "litellm-internal-health-check"
BATCH_TERMINAL_STATUSES: Final = frozenset({"completed", "failed", "cancelled", "expired"})
FAILED_BATCH_POLL_SECONDS: Final = 120.0
@ -281,6 +283,14 @@ class TestKeyAttribution:
"rust_control_plane",
],
)
@meta(
Subject(
domain=Domain.SPEND_BUDGETS,
route=Route.SPEND_REPORTING,
providers=(Provider.GEMINI, Provider.ANTHROPIC, Provider.OPENAI),
models=DRIVEN_MODELS,
)
)
def test_every_write_path_row_joins_the_key(self, client: SpendClient, driven: DrivenKey) -> None:
assert tuple(path.name for path in driven.paths) == WRITE_PATHS
found: Final = tuple((path, client.proxy.poll_logs_for_request_id(path.request_id)) for path in driven.paths)
@ -317,6 +327,14 @@ class TestKeyAttribution:
"rust_control_plane",
],
)
@meta(
Subject(
domain=Domain.SPEND_BUDGETS,
route=Route.SPEND_REPORTING,
providers=(Provider.GEMINI, Provider.ANTHROPIC, Provider.OPENAI),
models=DRIVEN_MODELS,
)
)
def test_spend_logs_by_key_return_every_row_with_the_alias(self, client: SpendClient, driven: DrivenKey) -> None:
expected_ids: Final = frozenset(path.request_id for path in driven.paths)
rows: Final = client.poll_logs_for_key(
@ -345,6 +363,14 @@ class TestKeyAttribution:
"rust_control_plane",
],
)
@meta(
Subject(
domain=Domain.SPEND_BUDGETS,
route=Route.SPEND_REPORTING,
providers=(Provider.GEMINI, Provider.ANTHROPIC, Provider.OPENAI),
models=DRIVEN_MODELS,
)
)
def test_user_daily_activity_reports_alias_and_email(self, client: SpendClient, driven: DrivenKey) -> None:
breakdown: Final[DailyActivityKeyBreakdown | None] = client.poll_daily_activity_for_key(
driven.identity.token,
@ -367,6 +393,14 @@ class TestKeyAttribution:
"quota_management.spend_tracking.key_attribution.health_rows_keep_service_account",
exercised_on=["chat_completions"],
)
@meta(
Subject(
domain=Domain.SPEND_BUDGETS,
route=Route.HEALTH,
providers=(Provider.GEMINI,),
models=(CHAT_MODEL,),
)
)
def test_health_check_rows_keep_the_service_account_key(self, client: SpendClient) -> None:
started_at: Final = datetime.now(timezone.utc)
probe: Final = client.health(CHAT_MODEL)
@ -380,6 +414,15 @@ class TestKeyAttribution:
"quota_management.spend_tracking.key_attribution.retrieve_batch_cost_joins_retrieving_key",
exercised_on=["batches"],
)
@meta(
Subject(
domain=Domain.SPEND_BUDGETS,
route=Route.BATCHES,
providers=(Provider.OPENAI,),
models=(BATCH_MODEL,),
mode=Mode.BATCH,
)
)
def test_terminal_batch_cost_row_joins_the_retrieving_key(self, client: SpendClient, driven: DrivenKey) -> None:
provider_batch_id: Final = _provider_batch_id(_driven_batch_id(driven))
fetched: Final = _await_terminal_batch(client, driven.identity.key, provider_batch_id)

View file

@ -14,6 +14,7 @@ write path are all still under test with zero provider calls.
import pytest
from e2e_config import CHEAP_OPENAI_MODEL, provider_edge_base
from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta
from lifecycle import ResourceManager
from models import LiteLLMParamsBody
from spend_e2e_client import SpendClient, unique_marker, unwrap
@ -22,6 +23,15 @@ pytestmark = [pytest.mark.e2e, pytest.mark.replayable]
@pytest.mark.covers("quota_management.spend_tracking.chat_completions.logs_cost")
@meta(
Subject(
domain=Domain.SPEND_BUDGETS,
route=Route.CHAT_COMPLETIONS,
providers=(Provider.OPENAI,),
models=(f"openai/{CHEAP_OPENAI_MODEL}",),
mode=Mode.NONSTREAM,
)
)
def test_edge_wired_chat_writes_nonzero_spend_row(
client: SpendClient, resources: ResourceManager, scoped_key: str
) -> None:

View file

@ -35,6 +35,7 @@ from cost_rows import (
)
from e2e_config import CHEAP_OPENAI_MODEL, unique_marker
from e2e_http import unwrap
from e2e_metadata import Capability, Domain, Mode, Provider, Subject, meta
from lifecycle import ResourceManager
from models import (
AnthropicMessagesBody,
@ -97,6 +98,15 @@ def _served_tier(chunks: list[_StreamChunk]) -> str:
class TestServiceTierPricing:
@pytest.mark.covers("quota_management.spend_tracking.service_tier.bills_tier_rates")
@meta(
Subject(
domain=Domain.SPEND_BUDGETS,
providers=(Provider.OPENAI,),
models=(BACKEND,),
capabilities=(Capability.REASONING,),
mode=Mode.NONSTREAM,
)
)
def test_priority_tier_bills_priority_rates(
self, client: SpendClient, resources: ResourceManager, scoped_key: str
) -> None:

View file

@ -17,11 +17,13 @@ fast: no batch-write wait, no provider calls.
"""
from datetime import datetime, timedelta, timezone
from types import MappingProxyType
from typing import Final
import pytest
from e2e_http import ProbeResult
from e2e_metadata import Domain, Route, Subject, meta
from models import DateRangeParams
from spend_e2e_client import SpendClient
@ -103,13 +105,38 @@ def _probe(client: SpendClient, route: str) -> ProbeResult:
return client.probe(route, params=_date_range())
@pytest.mark.parametrize("route", SPEND_ROUTES)
_LIST_ROUTES: Final = MappingProxyType(
{
"/key/list": Route.KEY_MANAGEMENT,
"/user/list": Route.USER_MANAGEMENT,
"/team/list": Route.TEAM_MANAGEMENT,
"/organization/list": Route.ORGANIZATION_MANAGEMENT,
"/customer/list": Route.CUSTOMER_MANAGEMENT,
}
)
_ROUTE_CASES: Final = tuple(
pytest.param(
path,
marks=meta(Subject(domain=Domain.SPEND_BUDGETS, route=_LIST_ROUTES.get(path, Route.SPEND_REPORTING))),
)
for path in SPEND_ROUTES
)
@pytest.mark.parametrize("route", _ROUTE_CASES)
def test_spend_route_responsive(client: SpendClient, route: str) -> None:
result = _probe(client, route)
print(f"{route} -> {result.status_code}\n{result.body[:600]}")
assert result.healthy, f"{route} -> {result.status_code}\n{result.body[:600]}"
@meta(
Subject(
domain=Domain.SPEND_BUDGETS,
route=Route.SPEND_REPORTING,
)
)
def test_schema_listed_spend_routes_are_responsive(client: SpendClient) -> None:
"""Probe any spend GET route the schema lists that isn't in SPEND_ROUTES."""
schema = client.openapi()

View file

@ -22,6 +22,7 @@ from typing import Final
import pytest
from e2e_http import RateLimitedError, Success
from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta
from lifecycle import ResourceManager
from models import KeyGenerateBody, LiteLLMParamsBody, SpendLogs, SpendLogsParams
from spend_e2e_client import (
@ -32,9 +33,16 @@ from spend_e2e_client import (
unique_marker,
unwrap,
)
from spend_reconciliation import BACKEND as TRAFFIC_BACKEND
pytestmark = pytest.mark.e2e
GEMINI_MODEL = "gemini-2.5-flash"
CLAUDE_MODEL = "claude-haiku-4-5"
CODEX_MODEL = "openai-responses-codex"
EMBEDDING_MODEL = "openai-text-embedding-3-small"
OPENAI_BACKEND = "openai/gpt-5.5"
def _approx_equal(actual: float, expected: float) -> bool:
"""Within 1% or 1e-9 absolute - spend math, not exact float identity."""
@ -70,13 +78,22 @@ def _require_row(
@pytest.mark.covers("quota_management.spend_tracking.chat_completions.logs_cost")
@meta(
Subject(
domain=Domain.SPEND_BUDGETS,
route=Route.CHAT_COMPLETIONS,
providers=(Provider.GEMINI,),
models=(GEMINI_MODEL,),
mode=Mode.NONSTREAM,
)
)
def test_chat_completion_writes_nonzero_spend_row(
client: SpendClient, scoped_key: str
) -> None:
chat = unwrap(
client.chat(
scoped_key,
"gemini-2.5-flash",
GEMINI_MODEL,
f"reply with one word {unique_marker()}",
max_tokens=16,
)
@ -90,7 +107,7 @@ def test_chat_completion_writes_nonzero_spend_row(
assert (row.spend or 0) > 0, f"chat row should cost > 0: {_summarize(rows)}"
assert row.status == "success"
assert row.cache_hit != "True", "fresh call must not be a cache hit"
assert "gemini-2.5-flash" in (row.model or "")
assert GEMINI_MODEL in (row.model or "")
prompt = row.prompt_tokens or 0
completion = row.completion_tokens or 0
@ -105,12 +122,21 @@ def test_chat_completion_writes_nonzero_spend_row(
@pytest.mark.covers("quota_management.spend_tracking.stream.logs_cost")
@meta(
Subject(
domain=Domain.SPEND_BUDGETS,
route=Route.CHAT_COMPLETIONS,
providers=(Provider.GEMINI,),
models=(GEMINI_MODEL,),
mode=Mode.STREAM,
)
)
def test_streaming_chat_completion_tracks_spend(
client: SpendClient, scoped_key: str
) -> None:
result = client.chat_stream(
scoped_key,
"gemini-2.5-flash",
GEMINI_MODEL,
f"count to three {unique_marker()}",
max_tokens=64,
)
@ -133,6 +159,15 @@ def test_streaming_chat_completion_tracks_spend(
@pytest.mark.covers("quota_management.spend_tracking.messages_bridge.logs_cost")
@meta(
Subject(
domain=Domain.SPEND_BUDGETS,
route=Route.MESSAGES,
providers=(Provider.OPENAI,),
models=(CODEX_MODEL,),
mode=Mode.STREAM,
)
)
def test_streaming_messages_via_responses_bridge_tracks_spend(
client: SpendClient, scoped_key: str
) -> None:
@ -150,7 +185,7 @@ def test_streaming_messages_via_responses_bridge_tracks_spend(
"""
result = client.messages_stream(
scoped_key,
"openai-responses-codex",
CODEX_MODEL,
f"reply with exactly one word {unique_marker()}",
max_tokens=64,
)
@ -203,13 +238,22 @@ def test_streaming_messages_via_responses_bridge_tracks_spend(
@pytest.mark.covers("quota_management.spend_tracking.embeddings.logs_cost")
@pytest.mark.covers("llm.embeddings.openai.basic.nonstream.cost_logged")
@meta(
Subject(
domain=Domain.SPEND_BUDGETS,
route=Route.EMBEDDINGS,
providers=(Provider.OPENAI,),
models=(EMBEDDING_MODEL,),
mode=Mode.NONSTREAM,
)
)
def test_embedding_writes_nonzero_spend_row(
client: SpendClient, scoped_key: str
) -> None:
_ = unwrap(
client.embed(
scoped_key,
"openai-text-embedding-3-small",
EMBEDDING_MODEL,
f"vectorize this sentence {unique_marker()}",
)
)
@ -226,6 +270,14 @@ def test_embedding_writes_nonzero_spend_row(
@pytest.mark.covers("quota_management.spend_tracking.cache_hit.zero_cost")
@meta(
Subject(
domain=Domain.SPEND_BUDGETS,
providers=(Provider.GEMINI,),
models=(GEMINI_MODEL,),
mode=Mode.NONSTREAM,
)
)
def test_cache_hit_is_zero_cost_and_suffixed(
client: SpendClient, scoped_key: str
) -> None:
@ -234,8 +286,8 @@ def test_cache_hit_is_zero_cost_and_suffixed(
# populated. The marker keeps each run isolated - a fixed prompt would persist
# in the shared response cache across runs and make both calls hit (flaky).
prompt = f"What is the capital of France? Answer in one word. {unique_marker()}"
_ = unwrap(client.chat(scoped_key, "gemini-2.5-flash", prompt, max_tokens=16, cache=None))
_ = unwrap(client.chat(scoped_key, "gemini-2.5-flash", prompt, max_tokens=16, cache=None))
_ = unwrap(client.chat(scoped_key, GEMINI_MODEL, prompt, max_tokens=16, cache=None))
_ = unwrap(client.chat(scoped_key, GEMINI_MODEL, prompt, max_tokens=16, cache=None))
rows = client.poll_logs_for_key(
scoped_key,
@ -262,12 +314,20 @@ def test_cache_hit_is_zero_cost_and_suffixed(
@pytest.mark.covers("quota_management.spend_tracking.key_rollup.matches_sum_of_logs")
@meta(
Subject(
domain=Domain.SPEND_BUDGETS,
providers=(Provider.GEMINI,),
models=(GEMINI_MODEL,),
mode=Mode.NONSTREAM,
)
)
def test_key_spend_equals_sum_of_logs(client: SpendClient, scoped_key: str) -> None:
for _ in range(2):
_ = unwrap(
client.chat(
scoped_key,
"gemini-2.5-flash",
GEMINI_MODEL,
f"say hi {unique_marker()}",
max_tokens=16,
)
@ -290,6 +350,14 @@ def test_key_spend_equals_sum_of_logs(client: SpendClient, scoped_key: str) -> N
@pytest.mark.replayable
@pytest.mark.covers("quota_management.spend_tracking.concurrent_burst.loses_no_spend")
@meta(
Subject(
domain=Domain.SPEND_BUDGETS,
providers=(Provider.OPENAI,),
models=(TRAFFIC_BACKEND,),
mode=Mode.NONSTREAM,
)
)
def test_burst_of_concurrent_calls_loses_no_spend(
client: SpendClient, resources: ResourceManager
) -> None:
@ -307,6 +375,15 @@ def test_burst_of_concurrent_calls_loses_no_spend(
@pytest.mark.covers("quota_management.spend_tracking.pagination.keeps_total")
@meta(
Subject(
domain=Domain.SPEND_BUDGETS,
route=Route.SPEND_REPORTING,
providers=(Provider.GEMINI,),
models=(GEMINI_MODEL,),
mode=Mode.NONSTREAM,
)
)
def test_spend_logs_v2_pagination_caps_pages_and_keeps_total(
client: SpendClient, scoped_key: str
) -> None:
@ -323,7 +400,7 @@ def test_spend_logs_v2_pagination_caps_pages_and_keeps_total(
_ = unwrap(
client.chat(
scoped_key,
"gemini-2.5-flash",
GEMINI_MODEL,
f"page fodder {unique_marker()}",
max_tokens=16,
)
@ -360,11 +437,19 @@ def test_spend_logs_v2_pagination_caps_pages_and_keeps_total(
@pytest.mark.covers("quota_management.spend_tracking.tags.attributes_spend")
@meta(
Subject(
domain=Domain.SPEND_BUDGETS,
providers=(Provider.GEMINI,),
models=(GEMINI_MODEL,),
mode=Mode.NONSTREAM,
)
)
def test_request_tags_round_trip(client: SpendClient, scoped_key: str) -> None:
tag = f"e2e-spend-{unique_marker()}"
_ = unwrap(
client.chat(
scoped_key, "gemini-2.5-flash", "tagged request", tags=[tag], max_tokens=16
scoped_key, GEMINI_MODEL, "tagged request", tags=[tag], max_tokens=16
)
)
@ -377,6 +462,14 @@ def test_request_tags_round_trip(client: SpendClient, scoped_key: str) -> None:
@pytest.mark.covers("quota_management.spend_tracking.tags.attributes_spend")
@meta(
Subject(
domain=Domain.SPEND_BUDGETS,
providers=(Provider.GEMINI,),
models=(GEMINI_MODEL,),
mode=Mode.NONSTREAM,
)
)
def test_tag_spend_matches_sum_of_tagged_logs(
client: SpendClient, scoped_key: str
) -> None:
@ -387,7 +480,7 @@ def test_tag_spend_matches_sum_of_tagged_logs(
_ = unwrap(
client.chat(
scoped_key,
"gemini-2.5-flash",
GEMINI_MODEL,
f"hi {unique_marker()}",
tags=[tag],
max_tokens=16,
@ -415,12 +508,20 @@ def test_tag_spend_matches_sum_of_tagged_logs(
@pytest.mark.covers("quota_management.spend_tracking.end_user.attributes_spend")
@meta(
Subject(
domain=Domain.SPEND_BUDGETS,
providers=(Provider.GEMINI,),
models=(GEMINI_MODEL,),
mode=Mode.NONSTREAM,
)
)
def test_end_user_spend_attributed_on_row(
client: SpendClient, scoped_key: str, resources: ResourceManager
) -> None:
customer = resources.customer(f"e2e-cust-{unique_marker()}")
_ = unwrap(
client.chat(scoped_key, "gemini-2.5-flash", "hi", user=customer, max_tokens=16)
client.chat(scoped_key, GEMINI_MODEL, "hi", user=customer, max_tokens=16)
)
rows = client.poll_logs_for_key(
@ -448,7 +549,7 @@ def test_end_user_header_attributes_responses_row(
{"authorization": f"Bearer {scoped_key}", header: customer, "x-litellm-tags": tag}
)
sent = client.send_responses_with_headers(
headers, "openai-responses-codex", f"one word {unique_marker()}"
headers, CODEX_MODEL, f"one word {unique_marker()}"
)
assert sent.ok, f"/v1/responses failed with {sent.status_code}: {sent.body[:300]}"
@ -468,6 +569,14 @@ def test_end_user_header_attributes_responses_row(
@pytest.mark.covers("quota_management.spend_tracking.per_model.writes_own_rows")
@meta(
Subject(
domain=Domain.SPEND_BUDGETS,
providers=(Provider.GEMINI, Provider.ANTHROPIC),
models=(GEMINI_MODEL, CLAUDE_MODEL),
mode=Mode.NONSTREAM,
)
)
def test_each_model_on_a_shared_key_gets_its_own_row(
client: SpendClient, scoped_key: str
) -> None:
@ -478,27 +587,27 @@ def test_each_model_on_a_shared_key_gets_its_own_row(
sibling deployment, or collapses both calls onto one request_id fails here."""
gemini = unwrap(
client.chat(
scoped_key, "gemini-2.5-flash", f"one word {unique_marker()}", max_tokens=16
scoped_key, GEMINI_MODEL, f"one word {unique_marker()}", max_tokens=16
)
)
claude = unwrap(
client.chat(
scoped_key, "claude-haiku-4-5", f"one word {unique_marker()}", max_tokens=16
scoped_key, CLAUDE_MODEL, f"one word {unique_marker()}", max_tokens=16
)
)
def both_models_costed(rows: list[SpendLogRow]) -> bool:
costed = [r.model or "" for r in rows if (r.spend or 0) > 0]
return any("gemini-2.5-flash" in m for m in costed) and any(
"claude-haiku-4-5" in m for m in costed
return any(GEMINI_MODEL in m for m in costed) and any(
CLAUDE_MODEL in m for m in costed
)
rows = client.poll_logs_for_key(scoped_key, min_rows=2, predicate=both_models_costed)
gemini_row = _require_row(
rows, lambda r: "gemini-2.5-flash" in (r.model or ""), "for the gemini call"
rows, lambda r: GEMINI_MODEL in (r.model or ""), "for the gemini call"
)
claude_row = _require_row(
rows, lambda r: "claude-haiku-4-5" in (r.model or ""), "for the claude call"
rows, lambda r: CLAUDE_MODEL in (r.model or ""), "for the claude call"
)
assert (gemini_row.spend or 0) > 0, f"gemini row should cost > 0: {_summarize(rows)}"
@ -517,13 +626,21 @@ def test_each_model_on_a_shared_key_gets_its_own_row(
@pytest.mark.covers("quota_management.spend_tracking.failure.writes_failure_row")
@meta(
Subject(
domain=Domain.SPEND_BUDGETS,
providers=(Provider.OPENAI,),
models=(OPENAI_BACKEND,),
mode=Mode.NONSTREAM,
)
)
def test_failure_call_writes_failure_status_row(
client: SpendClient, resources: ResourceManager, scoped_key: str
) -> None:
model = f"e2e-spend-failure-{unique_marker()}"
model_id = client.proxy.create_model(
model,
LiteLLMParamsBody(model="openai/gpt-5.5", api_key="sk-invalid-e2e-failure-row"),
LiteLLMParamsBody(model=OPENAI_BACKEND, api_key="sk-invalid-e2e-failure-row"),
)
resources.defer(lambda: client.proxy.delete_model(model_id))
@ -550,7 +667,7 @@ def test_failure_rows_share_normalized_error_across_provider_wording(
carries the same stable normalized_error cluster key."""
marker = unique_marker()
deployments: Final = (
(f"e2e-norm-openai-{marker}", "openai/gpt-5.5"),
(f"e2e-norm-openai-{marker}", OPENAI_BACKEND),
(f"e2e-norm-anthropic-{marker}", "anthropic/claude-haiku-4-5"),
)
for name, provider_model in deployments:
@ -593,7 +710,7 @@ def test_pre_call_rejection_row_attributes_provider_and_model_id(
can count it."""
model = f"e2e-spend-precall-{unique_marker()}"
model_id = client.proxy.create_model(
model, LiteLLMParamsBody(model="openai/gpt-5.5", api_key="os.environ/OPENAI_API_KEY")
model, LiteLLMParamsBody(model=OPENAI_BACKEND, api_key="os.environ/OPENAI_API_KEY")
)
resources.defer(lambda: client.proxy.delete_model(model_id))
key = client.proxy.generate_key(KeyGenerateBody(models=[model], rpm_limit=1))
@ -624,9 +741,17 @@ def test_pre_call_rejection_row_attributes_provider_and_model_id(
@pytest.mark.covers("quota_management.spend_tracking.spend_calculate.returns_cost")
@meta(
Subject(
domain=Domain.SPEND_BUDGETS,
route=Route.SPEND_REPORTING,
providers=(Provider.GEMINI,),
models=(GEMINI_MODEL,),
)
)
def test_spend_calculate_returns_nonzero_cost(client: SpendClient) -> None:
cost = client.calculate_spend(
"gemini-2.5-flash", "estimate the cost of this request"
GEMINI_MODEL, "estimate the cost of this request"
)
assert cost > 0, (
"/spend/calculate returned 0 for gemini-2.5-flash; "
@ -634,6 +759,15 @@ def test_spend_calculate_returns_nonzero_cost(client: SpendClient) -> None:
)
@meta(
Subject(
domain=Domain.SPEND_BUDGETS,
route=Route.SPEND_REPORTING,
providers=(Provider.GEMINI,),
models=(GEMINI_MODEL,),
mode=Mode.NONSTREAM,
)
)
def test_spend_logs_endpoint_returns_spend(
client: SpendClient, scoped_key: str
) -> None:
@ -644,7 +778,7 @@ def test_spend_logs_endpoint_returns_spend(
call's nonzero spend must surface before the deadline."""
unwrap(
client.chat(
scoped_key, "gemini-2.5-flash", f"spend logs {unique_marker()}", max_tokens=16
scoped_key, GEMINI_MODEL, f"spend logs {unique_marker()}", max_tokens=16
)
)

View file

@ -14,11 +14,12 @@ from typing import Final
import pytest
from e2e_http import ProbeResult
from e2e_metadata import Domain, Provider, Route, Subject, meta
from lifecycle import ResourceManager
from proxy_client import Converged, await_converged
from pydantic import BaseModel
from spend_e2e_client import SpendClient
from spend_reconciliation import TeamTraffic, assert_logs_match, create_traffic
from spend_reconciliation import BACKEND, TeamTraffic, assert_logs_match, create_traffic
pytestmark = pytest.mark.e2e
@ -82,6 +83,14 @@ def _probe(client: SpendClient, params: BaseModel) -> ProbeResult:
class TestTeamDailyActivity:
@pytest.mark.replayable
@pytest.mark.covers("mgmt.team.daily_activity.happy_path")
@meta(
Subject(
domain=Domain.SPEND_BUDGETS,
route=Route.SPEND_REPORTING,
providers=(Provider.OPENAI,),
models=(BACKEND,),
)
)
def test_valid_date_range_returns_results_and_metadata(
self, client: SpendClient, resources: ResourceManager
) -> None:
@ -199,6 +208,12 @@ class TestTeamDailyActivity:
assert empty.metadata.total_failed_requests == 0
@pytest.mark.covers("mgmt.team.daily_activity.missing_start_date_rejected")
@meta(
Subject(
domain=Domain.SPEND_BUDGETS,
route=Route.SPEND_REPORTING,
)
)
def test_missing_start_date_is_rejected(self, client: SpendClient) -> None:
end = datetime.now(timezone.utc).date().isoformat()
result = _probe(client, TeamDailyActivityParams(end_date=end, page=1))
@ -207,6 +222,12 @@ class TestTeamDailyActivity:
)
@pytest.mark.covers("mgmt.team.daily_activity.missing_end_date_rejected")
@meta(
Subject(
domain=Domain.SPEND_BUDGETS,
route=Route.SPEND_REPORTING,
)
)
def test_missing_end_date_is_rejected(self, client: SpendClient) -> None:
start = (datetime.now(timezone.utc).date() - timedelta(days=1)).isoformat()
result = _probe(client, TeamDailyActivityParams(start_date=start, page=1))