test(e2e): declare models through the constant each test drives
Some checks failed
LiteLLM Rust / rust-lint (push) Has been cancelled
LiteLLM Rust / rust-test (push) Has been cancelled
LiteLLM Rust / rust-wheel (push) Has been cancelled
Terraform Provider / gofmt, vet, build, test (push) Has been cancelled
Terraform Provider / Provider endpoints vs proxy OpenAPI schema (push) Has been cancelled

43 @meta declarations in quota_management typed the model name out again, so changing the call would leave the coverage report naming the old model. Each file now has one constant used by both, and a guard fails on any model written as a string literal in @meta
This commit is contained in:
ryan-crabbe-berri 2026-09-30 18:31:33 -07:00
parent e733cc1ffd
commit 391d6ab1f5
11 changed files with 124 additions and 78 deletions

View file

@ -272,6 +272,36 @@ class TestProviderMirrorsLitellm:
assert not unknown, f"not LlmProviders values: {unknown}"
E2E_DIR: Final = Path(__file__).resolve().parents[1] / "e2e"
def _hand_typed_models(path: Path) -> Iterator[str]:
for node in ast.walk(ast.parse(path.read_text())):
match node:
case ast.Call(func=ast.Name(id="Subject"), keywords=keywords):
for keyword in keywords:
match keyword:
case ast.keyword(arg="models", value=ast.Tuple(elts=models)):
yield from (
f"{path.relative_to(E2E_DIR)}:{model.lineno} {model.value!r}"
for model in models
if isinstance(model, ast.Constant)
)
case _:
pass
case _:
pass
def test_a_declared_model_names_the_constant_the_test_drives() -> None:
"""A model typed out in `@meta` is a second copy of the one the test calls, so
changing the call would leave the coverage report naming the old model."""
offenders: Final = tuple(
offender for path in sorted(E2E_DIR.rglob("*.py")) for offender in _hand_typed_models(path)
)
assert offenders == ()
class TestStepRecording:
"""`@step`-decorated harness helpers append to the running test's story as
they execute.

View file

@ -24,12 +24,13 @@ from lifecycle import ResourceManager
pytestmark = pytest.mark.e2e
MODEL = "claude-haiku-4-5"
TINY_CAP = 3e-6
ROOMY_CAP = 100.0
def _chat(client: BudgetClient, key: str, *, user: str | None = None) -> StreamingResponse:
return client.chat(key, "claude-haiku-4-5", f"spend {unique_marker()}", max_tokens=16, user=user)
return client.chat(key, MODEL, f"spend {unique_marker()}", max_tokens=16, user=user)
def _assert_budget_blocks(client: BudgetClient, key: str, *, user: str = "") -> StreamingResponse:
@ -62,7 +63,7 @@ class TestBudgetBlocksPerLevel:
domain=Domain.SPEND_BUDGETS,
route=Route.CHAT_COMPLETIONS,
providers=(Provider.ANTHROPIC,),
models=("claude-haiku-4-5",),
models=(MODEL,),
mode=Mode.NONSTREAM,
)
)
@ -78,7 +79,7 @@ class TestBudgetBlocksPerLevel:
domain=Domain.SPEND_BUDGETS,
route=Route.CHAT_COMPLETIONS,
providers=(Provider.ANTHROPIC,),
models=("claude-haiku-4-5",),
models=(MODEL,),
mode=Mode.NONSTREAM,
)
)
@ -103,7 +104,7 @@ class TestBudgetBlocksPerLevel:
domain=Domain.SPEND_BUDGETS,
route=Route.CHAT_COMPLETIONS,
providers=(Provider.ANTHROPIC,),
models=("claude-haiku-4-5",),
models=(MODEL,),
mode=Mode.NONSTREAM,
)
)
@ -146,7 +147,7 @@ class TestBudgetBlocksPerLevel:
domain=Domain.SPEND_BUDGETS,
route=Route.CHAT_COMPLETIONS,
providers=(Provider.ANTHROPIC,),
models=("claude-haiku-4-5",),
models=(MODEL,),
mode=Mode.NONSTREAM,
)
)
@ -156,7 +157,7 @@ class TestBudgetBlocksPerLevel:
customer = f"e2e-budget-cust-{unique_marker()}"
client.create_customer(customer, max_budget=TINY_CAP)
resources.defer(lambda: client.delete_customers([customer]))
key = client.generate_key(models=["claude-haiku-4-5"])
key = client.generate_key(models=[MODEL])
resources.defer(lambda: client.delete_key(key))
_assert_budget_blocks(client, key, user=customer)
@ -167,7 +168,7 @@ class TestBudgetBlocksPerLevel:
domain=Domain.SPEND_BUDGETS,
route=Route.CHAT_COMPLETIONS,
providers=(Provider.ANTHROPIC,),
models=("claude-haiku-4-5",),
models=(MODEL,),
mode=Mode.NONSTREAM,
)
)
@ -190,7 +191,7 @@ class TestBudgetBlocksPerLevel:
domain=Domain.SPEND_BUDGETS,
route=Route.CHAT_COMPLETIONS,
providers=(Provider.ANTHROPIC,),
models=("claude-haiku-4-5",),
models=(MODEL,),
mode=Mode.NONSTREAM,
)
)
@ -226,7 +227,7 @@ class TestKeyBudgetBlocksAcrossKeyKinds:
domain=Domain.SPEND_BUDGETS,
route=Route.CHAT_COMPLETIONS,
providers=(Provider.ANTHROPIC,),
models=("claude-haiku-4-5",),
models=(MODEL,),
mode=Mode.NONSTREAM,
)
)
@ -249,7 +250,7 @@ class TestKeyBudgetBlocksAcrossKeyKinds:
domain=Domain.SPEND_BUDGETS,
route=Route.CHAT_COMPLETIONS,
providers=(Provider.ANTHROPIC,),
models=("claude-haiku-4-5",),
models=(MODEL,),
mode=Mode.NONSTREAM,
)
)
@ -270,7 +271,7 @@ class TestKeyBudgetBlocksAcrossKeyKinds:
domain=Domain.SPEND_BUDGETS,
route=Route.CHAT_COMPLETIONS,
providers=(Provider.ANTHROPIC,),
models=("claude-haiku-4-5",),
models=(MODEL,),
mode=Mode.NONSTREAM,
)
)

View file

@ -28,6 +28,7 @@ from models import BudgetWindow
pytestmark = pytest.mark.e2e
MODEL = "claude-haiku-4-5"
WINDOW_SECONDS = 30
RESET_DEADLINE_SECONDS = 150
TINY_CAP = 3e-6
@ -35,7 +36,7 @@ SPEND_SETTLE_DEADLINE_SECONDS = 90
def _call(client: BudgetClient, key: str):
return client.chat(key, "claude-haiku-4-5", f"advance {unique_marker()}", max_tokens=16)
return client.chat(key, MODEL, f"advance {unique_marker()}", max_tokens=16)
def _poll_key_spend(client: BudgetClient, key: str, settled: Callable[[float], bool], problem: str) -> None:
@ -98,7 +99,7 @@ def test_key_with_budget_duration_schedules_reset_at_creation(client: BudgetClie
domain=Domain.SPEND_BUDGETS,
route=Route.CHAT_COMPLETIONS,
providers=(Provider.ANTHROPIC,),
models=("claude-haiku-4-5",),
models=(MODEL,),
mode=Mode.NONSTREAM,
)
)
@ -124,7 +125,7 @@ def test_key_spend_blocks_at_cap(client: BudgetClient, resources: ResourceManage
domain=Domain.SPEND_BUDGETS,
route=Route.CHAT_COMPLETIONS,
providers=(Provider.ANTHROPIC,),
models=("claude-haiku-4-5",),
models=(MODEL,),
mode=Mode.NONSTREAM,
)
)
@ -169,7 +170,7 @@ def test_key_budget_reset_at_advances_after_window(client: BudgetClient, resourc
domain=Domain.SPEND_BUDGETS,
route=Route.CHAT_COMPLETIONS,
providers=(Provider.ANTHROPIC,),
models=("claude-haiku-4-5",),
models=(MODEL,),
mode=Mode.NONSTREAM,
)
)
@ -222,7 +223,7 @@ def test_multi_window_key_resets_each_window_independently(client: BudgetClient,
domain=Domain.SPEND_BUDGETS,
route=Route.CHAT_COMPLETIONS,
providers=(Provider.ANTHROPIC,),
models=("claude-haiku-4-5",),
models=(MODEL,),
mode=Mode.NONSTREAM,
)
)
@ -264,7 +265,7 @@ def test_team_member_budget_reset_at_advances(client: BudgetClient, resources: R
domain=Domain.SPEND_BUDGETS,
route=Route.CHAT_COMPLETIONS,
providers=(Provider.ANTHROPIC,),
models=("claude-haiku-4-5",),
models=(MODEL,),
mode=Mode.NONSTREAM,
)
)

View file

@ -12,6 +12,7 @@ from lifecycle import ResourceManager
pytestmark = pytest.mark.e2e
MODEL = "claude-haiku-4-5"
TINY_CAP = 3e-6
ROOMY_CAP = 100.0
WINDOW = "30s"
@ -19,7 +20,7 @@ RESET_DEADLINE_SECONDS = 150
def _call(client: BudgetClient, key: str):
return client.chat(key, "claude-haiku-4-5", f"reset {unique_marker()}", max_tokens=16)
return client.chat(key, MODEL, f"reset {unique_marker()}", max_tokens=16)
def _drive_to_block(client: BudgetClient, key: str) -> None:
@ -55,7 +56,7 @@ class TestBudgetResetPerLevel:
domain=Domain.SPEND_BUDGETS,
route=Route.CHAT_COMPLETIONS,
providers=(Provider.ANTHROPIC,),
models=("claude-haiku-4-5",),
models=(MODEL,),
mode=Mode.NONSTREAM,
)
)
@ -72,7 +73,7 @@ class TestBudgetResetPerLevel:
domain=Domain.SPEND_BUDGETS,
route=Route.CHAT_COMPLETIONS,
providers=(Provider.ANTHROPIC,),
models=("claude-haiku-4-5",),
models=(MODEL,),
mode=Mode.NONSTREAM,
)
)
@ -93,7 +94,7 @@ class TestBudgetResetPerLevel:
domain=Domain.SPEND_BUDGETS,
route=Route.CHAT_COMPLETIONS,
providers=(Provider.ANTHROPIC,),
models=("claude-haiku-4-5",),
models=(MODEL,),
mode=Mode.NONSTREAM,
)
)
@ -124,7 +125,7 @@ class TestBudgetResetPerLevel:
domain=Domain.SPEND_BUDGETS,
route=Route.CHAT_COMPLETIONS,
providers=(Provider.ANTHROPIC,),
models=("claude-haiku-4-5",),
models=(MODEL,),
mode=Mode.NONSTREAM,
)
)
@ -151,7 +152,7 @@ class TestKeyBudgetResetAcrossKeyKinds:
domain=Domain.SPEND_BUDGETS,
route=Route.CHAT_COMPLETIONS,
providers=(Provider.ANTHROPIC,),
models=("claude-haiku-4-5",),
models=(MODEL,),
mode=Mode.NONSTREAM,
)
)
@ -170,7 +171,7 @@ class TestKeyBudgetResetAcrossKeyKinds:
domain=Domain.SPEND_BUDGETS,
route=Route.CHAT_COMPLETIONS,
providers=(Provider.ANTHROPIC,),
models=("claude-haiku-4-5",),
models=(MODEL,),
mode=Mode.NONSTREAM,
)
)
@ -189,7 +190,7 @@ class TestKeyBudgetResetAcrossKeyKinds:
domain=Domain.SPEND_BUDGETS,
route=Route.CHAT_COMPLETIONS,
providers=(Provider.ANTHROPIC,),
models=("claude-haiku-4-5",),
models=(MODEL,),
mode=Mode.NONSTREAM,
)
)

View file

@ -17,6 +17,8 @@ from lifecycle import ResourceManager
pytestmark = pytest.mark.e2e
MODEL = "claude-haiku-4-5"
@pytest.mark.covers("quota_management.budget.soft.alerts_without_blocking")
@meta(
@ -24,7 +26,7 @@ pytestmark = pytest.mark.e2e
domain=Domain.SPEND_BUDGETS,
route=Route.CHAT_COMPLETIONS,
providers=(Provider.ANTHROPIC,),
models=("claude-haiku-4-5",),
models=(MODEL,),
mode=Mode.NONSTREAM,
)
)
@ -37,7 +39,7 @@ def test_soft_budget_does_not_block(
for _ in range(3):
result = client.chat(
key, "claude-haiku-4-5", f"hi {unique_marker()}", max_tokens=16
key, MODEL, f"hi {unique_marker()}", max_tokens=16
)
assert not is_budget_block(result), (
"soft_budget blocked a request; it must alert only, not block "

View file

@ -18,13 +18,14 @@ from lifecycle import ResourceManager
pytestmark = pytest.mark.e2e
MODEL = "claude-haiku-4-5"
TINY_BUDGET = 1e-6
def _tagged_call(client: BudgetClient, key: str, tag: str):
result = client.chat(
key,
"claude-haiku-4-5",
MODEL,
f"hi {unique_marker()}",
tags=[tag],
max_tokens=64,
@ -40,7 +41,7 @@ def _tagged_call(client: BudgetClient, key: str, tag: str):
domain=Domain.SPEND_BUDGETS,
route=Route.CHAT_COMPLETIONS,
providers=(Provider.ANTHROPIC,),
models=("claude-haiku-4-5",),
models=(MODEL,),
mode=Mode.NONSTREAM,
)
)

View file

@ -11,6 +11,7 @@ from lifecycle import ResourceManager
pytestmark = pytest.mark.e2e
MODEL = "claude-haiku-4-5"
MEMBER_BUDGET = 1.0 # default member budget is $50, we're testing with a smaller value
def _as_datetime(value: str) -> datetime:
@ -23,7 +24,7 @@ def _as_datetime(value: str) -> datetime:
domain=Domain.SPEND_BUDGETS,
route=Route.CHAT_COMPLETIONS,
providers=(Provider.ANTHROPIC,),
models=("claude-haiku-4-5",),
models=(MODEL,),
mode=Mode.NONSTREAM,
)
)
@ -44,7 +45,7 @@ def test_team_member_budget_reset_keeps_advancing(client: BudgetClient, resource
# the member can spend within the team while the window is live
key = client.generate_key(team_id=team_id, user_id=user_id)
resources.defer(lambda: client.delete_key(key))
require_successful_call(client.chat(key, "claude-haiku-4-5", f"reset {unique_marker()}", max_tokens=16))
require_successful_call(client.chat(key, MODEL, f"reset {unique_marker()}", max_tokens=16))
# once the window elapses the reset job must move budget_reset_at forward; a job
# that skips the member's budget row (the #25109 regression) leaves it pinned at

View file

@ -30,6 +30,7 @@ from models import BudgetWindow
pytestmark = pytest.mark.e2e
MODEL = "claude-haiku-4-5"
WINDOW_SECONDS = 30
SHORT_WINDOW = f"{WINDOW_SECONDS}s"
LONG_WINDOW = "1d"
@ -39,7 +40,7 @@ RESET_DEADLINE_SECONDS = 150
def _call(client: BudgetClient, key: str):
return client.chat(key, "claude-haiku-4-5", f"team-window {unique_marker()}", max_tokens=16)
return client.chat(key, MODEL, f"team-window {unique_marker()}", max_tokens=16)
def _drive_to_block(client: BudgetClient, key: str) -> StreamingResponse:
@ -58,7 +59,7 @@ def _drive_to_block(client: BudgetClient, key: str) -> StreamingResponse:
domain=Domain.SPEND_BUDGETS,
route=Route.CHAT_COMPLETIONS,
providers=(Provider.ANTHROPIC,),
models=("claude-haiku-4-5",),
models=(MODEL,),
mode=Mode.NONSTREAM,
)
)
@ -71,7 +72,7 @@ def test_team_short_window_blocks_then_resets(client: BudgetClient, resources: R
],
)
resources.defer(lambda: client.delete_team(team_id))
key = client.generate_key(team_id=team_id, models=["claude-haiku-4-5"])
key = client.generate_key(team_id=team_id, models=[MODEL])
resources.defer(lambda: client.delete_key(key))
# 1. exhaust the tight window -> litellm returns budget_exceeded
@ -100,7 +101,7 @@ def test_team_short_window_blocks_then_resets(client: BudgetClient, resources: R
domain=Domain.SPEND_BUDGETS,
route=Route.CHAT_COMPLETIONS,
providers=(Provider.ANTHROPIC,),
models=("claude-haiku-4-5",),
models=(MODEL,),
mode=Mode.NONSTREAM,
)
)
@ -115,7 +116,7 @@ def test_team_long_window_blocks_after_short_window_resets(client: BudgetClient,
],
)
resources.defer(lambda: client.delete_team(team_id))
key = client.generate_key(team_id=team_id, models=["claude-haiku-4-5"])
key = client.generate_key(team_id=team_id, models=[MODEL])
resources.defer(lambda: client.delete_key(key))
# 1. drive the key to being blocked, assert its blocked by budget budget_exceeded

View file

@ -9,6 +9,7 @@ from lifecycle import ResourceManager
from models import ChatBody, ChatMessage, ChatResponse, KeyGenerateBody, LiteLLMParamsBody, TeamNewBody
from spend_e2e_client import SpendClient
BACKEND: Final = "openai/gpt-5.6-luna"
INPUT_RATE: Final = 0.00004
OUTPUT_RATE: Final = 0.00008
@ -38,7 +39,7 @@ def create_traffic(client: SpendClient, resources: ResourceManager) -> tuple[Tea
model_id: Final = client.proxy.create_model(
model,
LiteLLMParamsBody(
model="openai/gpt-5.6-luna",
model=BACKEND,
api_key="os.environ/OPENAI_API_KEY",
api_base=None if base is None else f"{base}/v1",
input_cost_per_token=INPUT_RATE,

View file

@ -33,9 +33,16 @@ from spend_e2e_client import (
unique_marker,
unwrap,
)
from spend_reconciliation import BACKEND as TRAFFIC_BACKEND
pytestmark = pytest.mark.e2e
GEMINI_MODEL = "gemini-2.5-flash"
CLAUDE_MODEL = "claude-haiku-4-5"
CODEX_MODEL = "openai-responses-codex"
EMBEDDING_MODEL = "openai-text-embedding-3-small"
OPENAI_BACKEND = "openai/gpt-5.5"
def _approx_equal(actual: float, expected: float) -> bool:
"""Within 1% or 1e-9 absolute - spend math, not exact float identity."""
@ -76,7 +83,7 @@ def _require_row(
domain=Domain.SPEND_BUDGETS,
route=Route.CHAT_COMPLETIONS,
providers=(Provider.GEMINI,),
models=("gemini-2.5-flash",),
models=(GEMINI_MODEL,),
mode=Mode.NONSTREAM,
)
)
@ -86,7 +93,7 @@ def test_chat_completion_writes_nonzero_spend_row(
chat = unwrap(
client.chat(
scoped_key,
"gemini-2.5-flash",
GEMINI_MODEL,
f"reply with one word {unique_marker()}",
max_tokens=16,
)
@ -100,7 +107,7 @@ def test_chat_completion_writes_nonzero_spend_row(
assert (row.spend or 0) > 0, f"chat row should cost > 0: {_summarize(rows)}"
assert row.status == "success"
assert row.cache_hit != "True", "fresh call must not be a cache hit"
assert "gemini-2.5-flash" in (row.model or "")
assert GEMINI_MODEL in (row.model or "")
prompt = row.prompt_tokens or 0
completion = row.completion_tokens or 0
@ -120,7 +127,7 @@ def test_chat_completion_writes_nonzero_spend_row(
domain=Domain.SPEND_BUDGETS,
route=Route.CHAT_COMPLETIONS,
providers=(Provider.GEMINI,),
models=("gemini-2.5-flash",),
models=(GEMINI_MODEL,),
mode=Mode.STREAM,
)
)
@ -129,7 +136,7 @@ def test_streaming_chat_completion_tracks_spend(
) -> None:
result = client.chat_stream(
scoped_key,
"gemini-2.5-flash",
GEMINI_MODEL,
f"count to three {unique_marker()}",
max_tokens=64,
)
@ -157,7 +164,7 @@ def test_streaming_chat_completion_tracks_spend(
domain=Domain.SPEND_BUDGETS,
route=Route.MESSAGES,
providers=(Provider.OPENAI,),
models=("openai-responses-codex",),
models=(CODEX_MODEL,),
mode=Mode.STREAM,
)
)
@ -178,7 +185,7 @@ def test_streaming_messages_via_responses_bridge_tracks_spend(
"""
result = client.messages_stream(
scoped_key,
"openai-responses-codex",
CODEX_MODEL,
f"reply with exactly one word {unique_marker()}",
max_tokens=64,
)
@ -236,7 +243,7 @@ def test_streaming_messages_via_responses_bridge_tracks_spend(
domain=Domain.SPEND_BUDGETS,
route=Route.EMBEDDINGS,
providers=(Provider.OPENAI,),
models=("openai-text-embedding-3-small",),
models=(EMBEDDING_MODEL,),
mode=Mode.NONSTREAM,
)
)
@ -246,7 +253,7 @@ def test_embedding_writes_nonzero_spend_row(
_ = unwrap(
client.embed(
scoped_key,
"openai-text-embedding-3-small",
EMBEDDING_MODEL,
f"vectorize this sentence {unique_marker()}",
)
)
@ -268,7 +275,7 @@ def test_embedding_writes_nonzero_spend_row(
domain=Domain.SPEND_BUDGETS,
route=Route.CHAT_COMPLETIONS,
providers=(Provider.GEMINI,),
models=("gemini-2.5-flash",),
models=(GEMINI_MODEL,),
mode=Mode.NONSTREAM,
)
)
@ -280,8 +287,8 @@ def test_cache_hit_is_zero_cost_and_suffixed(
# populated. The marker keeps each run isolated - a fixed prompt would persist
# in the shared response cache across runs and make both calls hit (flaky).
prompt = f"What is the capital of France? Answer in one word. {unique_marker()}"
_ = unwrap(client.chat(scoped_key, "gemini-2.5-flash", prompt, max_tokens=16, cache=None))
_ = unwrap(client.chat(scoped_key, "gemini-2.5-flash", prompt, max_tokens=16, cache=None))
_ = unwrap(client.chat(scoped_key, GEMINI_MODEL, prompt, max_tokens=16, cache=None))
_ = unwrap(client.chat(scoped_key, GEMINI_MODEL, prompt, max_tokens=16, cache=None))
rows = client.poll_logs_for_key(
scoped_key,
@ -313,7 +320,7 @@ def test_cache_hit_is_zero_cost_and_suffixed(
domain=Domain.SPEND_BUDGETS,
route=Route.CHAT_COMPLETIONS,
providers=(Provider.GEMINI,),
models=("gemini-2.5-flash",),
models=(GEMINI_MODEL,),
mode=Mode.NONSTREAM,
)
)
@ -322,7 +329,7 @@ def test_key_spend_equals_sum_of_logs(client: SpendClient, scoped_key: str) -> N
_ = unwrap(
client.chat(
scoped_key,
"gemini-2.5-flash",
GEMINI_MODEL,
f"say hi {unique_marker()}",
max_tokens=16,
)
@ -350,7 +357,7 @@ def test_key_spend_equals_sum_of_logs(client: SpendClient, scoped_key: str) -> N
domain=Domain.SPEND_BUDGETS,
route=Route.CHAT_COMPLETIONS,
providers=(Provider.OPENAI,),
models=("openai/gpt-5.6-luna",),
models=(TRAFFIC_BACKEND,),
mode=Mode.NONSTREAM,
)
)
@ -376,7 +383,7 @@ def test_burst_of_concurrent_calls_loses_no_spend(
domain=Domain.SPEND_BUDGETS,
route=Route.CHAT_COMPLETIONS,
providers=(Provider.GEMINI,),
models=("gemini-2.5-flash",),
models=(GEMINI_MODEL,),
mode=Mode.NONSTREAM,
)
)
@ -396,7 +403,7 @@ def test_spend_logs_v2_pagination_caps_pages_and_keeps_total(
_ = unwrap(
client.chat(
scoped_key,
"gemini-2.5-flash",
GEMINI_MODEL,
f"page fodder {unique_marker()}",
max_tokens=16,
)
@ -438,7 +445,7 @@ def test_spend_logs_v2_pagination_caps_pages_and_keeps_total(
domain=Domain.SPEND_BUDGETS,
route=Route.CHAT_COMPLETIONS,
providers=(Provider.GEMINI,),
models=("gemini-2.5-flash",),
models=(GEMINI_MODEL,),
mode=Mode.NONSTREAM,
)
)
@ -446,7 +453,7 @@ def test_request_tags_round_trip(client: SpendClient, scoped_key: str) -> None:
tag = f"e2e-spend-{unique_marker()}"
_ = unwrap(
client.chat(
scoped_key, "gemini-2.5-flash", "tagged request", tags=[tag], max_tokens=16
scoped_key, GEMINI_MODEL, "tagged request", tags=[tag], max_tokens=16
)
)
@ -464,7 +471,7 @@ def test_request_tags_round_trip(client: SpendClient, scoped_key: str) -> None:
domain=Domain.SPEND_BUDGETS,
route=Route.CHAT_COMPLETIONS,
providers=(Provider.GEMINI,),
models=("gemini-2.5-flash",),
models=(GEMINI_MODEL,),
mode=Mode.NONSTREAM,
)
)
@ -478,7 +485,7 @@ def test_tag_spend_matches_sum_of_tagged_logs(
_ = unwrap(
client.chat(
scoped_key,
"gemini-2.5-flash",
GEMINI_MODEL,
f"hi {unique_marker()}",
tags=[tag],
max_tokens=16,
@ -511,7 +518,7 @@ def test_tag_spend_matches_sum_of_tagged_logs(
domain=Domain.SPEND_BUDGETS,
route=Route.CHAT_COMPLETIONS,
providers=(Provider.GEMINI,),
models=("gemini-2.5-flash",),
models=(GEMINI_MODEL,),
mode=Mode.NONSTREAM,
)
)
@ -520,7 +527,7 @@ def test_end_user_spend_attributed_on_row(
) -> None:
customer = resources.customer(f"e2e-cust-{unique_marker()}")
_ = unwrap(
client.chat(scoped_key, "gemini-2.5-flash", "hi", user=customer, max_tokens=16)
client.chat(scoped_key, GEMINI_MODEL, "hi", user=customer, max_tokens=16)
)
rows = client.poll_logs_for_key(
@ -548,7 +555,7 @@ def test_end_user_header_attributes_responses_row(
{"authorization": f"Bearer {scoped_key}", header: customer, "x-litellm-tags": tag}
)
sent = client.send_responses_with_headers(
headers, "openai-responses-codex", f"one word {unique_marker()}"
headers, CODEX_MODEL, f"one word {unique_marker()}"
)
assert sent.ok, f"/v1/responses failed with {sent.status_code}: {sent.body[:300]}"
@ -573,7 +580,7 @@ def test_end_user_header_attributes_responses_row(
domain=Domain.SPEND_BUDGETS,
route=Route.CHAT_COMPLETIONS,
providers=(Provider.GEMINI, Provider.ANTHROPIC),
models=("gemini-2.5-flash", "claude-haiku-4-5"),
models=(GEMINI_MODEL, CLAUDE_MODEL),
mode=Mode.NONSTREAM,
)
)
@ -587,27 +594,27 @@ def test_each_model_on_a_shared_key_gets_its_own_row(
sibling deployment, or collapses both calls onto one request_id fails here."""
gemini = unwrap(
client.chat(
scoped_key, "gemini-2.5-flash", f"one word {unique_marker()}", max_tokens=16
scoped_key, GEMINI_MODEL, f"one word {unique_marker()}", max_tokens=16
)
)
claude = unwrap(
client.chat(
scoped_key, "claude-haiku-4-5", f"one word {unique_marker()}", max_tokens=16
scoped_key, CLAUDE_MODEL, f"one word {unique_marker()}", max_tokens=16
)
)
def both_models_costed(rows: list[SpendLogRow]) -> bool:
costed = [r.model or "" for r in rows if (r.spend or 0) > 0]
return any("gemini-2.5-flash" in m for m in costed) and any(
"claude-haiku-4-5" in m for m in costed
return any(GEMINI_MODEL in m for m in costed) and any(
CLAUDE_MODEL in m for m in costed
)
rows = client.poll_logs_for_key(scoped_key, min_rows=2, predicate=both_models_costed)
gemini_row = _require_row(
rows, lambda r: "gemini-2.5-flash" in (r.model or ""), "for the gemini call"
rows, lambda r: GEMINI_MODEL in (r.model or ""), "for the gemini call"
)
claude_row = _require_row(
rows, lambda r: "claude-haiku-4-5" in (r.model or ""), "for the claude call"
rows, lambda r: CLAUDE_MODEL in (r.model or ""), "for the claude call"
)
assert (gemini_row.spend or 0) > 0, f"gemini row should cost > 0: {_summarize(rows)}"
@ -631,7 +638,7 @@ def test_each_model_on_a_shared_key_gets_its_own_row(
domain=Domain.SPEND_BUDGETS,
route=Route.CHAT_COMPLETIONS,
providers=(Provider.OPENAI,),
models=("openai/gpt-5.5",),
models=(OPENAI_BACKEND,),
mode=Mode.NONSTREAM,
)
)
@ -641,7 +648,7 @@ def test_failure_call_writes_failure_status_row(
model = f"e2e-spend-failure-{unique_marker()}"
model_id = client.proxy.create_model(
model,
LiteLLMParamsBody(model="openai/gpt-5.5", api_key="sk-invalid-e2e-failure-row"),
LiteLLMParamsBody(model=OPENAI_BACKEND, api_key="sk-invalid-e2e-failure-row"),
)
resources.defer(lambda: client.proxy.delete_model(model_id))
@ -668,7 +675,7 @@ def test_failure_rows_share_normalized_error_across_provider_wording(
carries the same stable normalized_error cluster key."""
marker = unique_marker()
deployments: Final = (
(f"e2e-norm-openai-{marker}", "openai/gpt-5.5"),
(f"e2e-norm-openai-{marker}", OPENAI_BACKEND),
(f"e2e-norm-anthropic-{marker}", "anthropic/claude-haiku-4-5"),
)
for name, provider_model in deployments:
@ -711,7 +718,7 @@ def test_pre_call_rejection_row_attributes_provider_and_model_id(
can count it."""
model = f"e2e-spend-precall-{unique_marker()}"
model_id = client.proxy.create_model(
model, LiteLLMParamsBody(model="openai/gpt-5.5", api_key="os.environ/OPENAI_API_KEY")
model, LiteLLMParamsBody(model=OPENAI_BACKEND, api_key="os.environ/OPENAI_API_KEY")
)
resources.defer(lambda: client.proxy.delete_model(model_id))
key = client.proxy.generate_key(KeyGenerateBody(models=[model], rpm_limit=1))
@ -747,12 +754,12 @@ def test_pre_call_rejection_row_attributes_provider_and_model_id(
domain=Domain.SPEND_BUDGETS,
route=Route.SPEND_REPORTING,
providers=(Provider.GEMINI,),
models=("gemini-2.5-flash",),
models=(GEMINI_MODEL,),
)
)
def test_spend_calculate_returns_nonzero_cost(client: SpendClient) -> None:
cost = client.calculate_spend(
"gemini-2.5-flash", "estimate the cost of this request"
GEMINI_MODEL, "estimate the cost of this request"
)
assert cost > 0, (
"/spend/calculate returned 0 for gemini-2.5-flash; "
@ -765,7 +772,7 @@ def test_spend_calculate_returns_nonzero_cost(client: SpendClient) -> None:
domain=Domain.SPEND_BUDGETS,
route=Route.CHAT_COMPLETIONS,
providers=(Provider.GEMINI,),
models=("gemini-2.5-flash",),
models=(GEMINI_MODEL,),
mode=Mode.NONSTREAM,
)
)
@ -779,7 +786,7 @@ def test_spend_logs_endpoint_returns_spend(
call's nonzero spend must surface before the deadline."""
unwrap(
client.chat(
scoped_key, "gemini-2.5-flash", f"spend logs {unique_marker()}", max_tokens=16
scoped_key, GEMINI_MODEL, f"spend logs {unique_marker()}", max_tokens=16
)
)

View file

@ -19,7 +19,7 @@ from lifecycle import ResourceManager
from proxy_client import Converged, await_converged
from pydantic import BaseModel
from spend_e2e_client import SpendClient
from spend_reconciliation import TeamTraffic, assert_logs_match, create_traffic
from spend_reconciliation import BACKEND, TeamTraffic, assert_logs_match, create_traffic
pytestmark = pytest.mark.e2e
@ -88,7 +88,7 @@ class TestTeamDailyActivity:
domain=Domain.SPEND_BUDGETS,
route=Route.SPEND_REPORTING,
providers=(Provider.OPENAI,),
models=("openai/gpt-5.6-luna",),
models=(BACKEND,),
)
)
def test_valid_date_range_returns_results_and_metadata(