mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-05 02:41:56 +00:00
test(e2e): declare models through the constant each test drives
Some checks failed
LiteLLM Rust / rust-lint (push) Has been cancelled
LiteLLM Rust / rust-test (push) Has been cancelled
LiteLLM Rust / rust-wheel (push) Has been cancelled
Terraform Provider / gofmt, vet, build, test (push) Has been cancelled
Terraform Provider / Provider endpoints vs proxy OpenAPI schema (push) Has been cancelled
Some checks failed
LiteLLM Rust / rust-lint (push) Has been cancelled
LiteLLM Rust / rust-test (push) Has been cancelled
LiteLLM Rust / rust-wheel (push) Has been cancelled
Terraform Provider / gofmt, vet, build, test (push) Has been cancelled
Terraform Provider / Provider endpoints vs proxy OpenAPI schema (push) Has been cancelled
43 @meta declarations in quota_management typed the model name out again, so changing the call would leave the coverage report naming the old model. Each file now has one constant used by both, and a guard fails on any model written as a string literal in @meta
This commit is contained in:
parent
e733cc1ffd
commit
391d6ab1f5
11 changed files with 124 additions and 78 deletions
|
|
@ -272,6 +272,36 @@ class TestProviderMirrorsLitellm:
|
|||
assert not unknown, f"not LlmProviders values: {unknown}"
|
||||
|
||||
|
||||
E2E_DIR: Final = Path(__file__).resolve().parents[1] / "e2e"
|
||||
|
||||
|
||||
def _hand_typed_models(path: Path) -> Iterator[str]:
|
||||
for node in ast.walk(ast.parse(path.read_text())):
|
||||
match node:
|
||||
case ast.Call(func=ast.Name(id="Subject"), keywords=keywords):
|
||||
for keyword in keywords:
|
||||
match keyword:
|
||||
case ast.keyword(arg="models", value=ast.Tuple(elts=models)):
|
||||
yield from (
|
||||
f"{path.relative_to(E2E_DIR)}:{model.lineno} {model.value!r}"
|
||||
for model in models
|
||||
if isinstance(model, ast.Constant)
|
||||
)
|
||||
case _:
|
||||
pass
|
||||
case _:
|
||||
pass
|
||||
|
||||
|
||||
def test_a_declared_model_names_the_constant_the_test_drives() -> None:
|
||||
"""A model typed out in `@meta` is a second copy of the one the test calls, so
|
||||
changing the call would leave the coverage report naming the old model."""
|
||||
offenders: Final = tuple(
|
||||
offender for path in sorted(E2E_DIR.rglob("*.py")) for offender in _hand_typed_models(path)
|
||||
)
|
||||
assert offenders == ()
|
||||
|
||||
|
||||
class TestStepRecording:
|
||||
"""`@step`-decorated harness helpers append to the running test's story as
|
||||
they execute.
|
||||
|
|
|
|||
|
|
@ -24,12 +24,13 @@ from lifecycle import ResourceManager
|
|||
|
||||
pytestmark = pytest.mark.e2e
|
||||
|
||||
MODEL = "claude-haiku-4-5"
|
||||
TINY_CAP = 3e-6
|
||||
ROOMY_CAP = 100.0
|
||||
|
||||
|
||||
def _chat(client: BudgetClient, key: str, *, user: str | None = None) -> StreamingResponse:
|
||||
return client.chat(key, "claude-haiku-4-5", f"spend {unique_marker()}", max_tokens=16, user=user)
|
||||
return client.chat(key, MODEL, f"spend {unique_marker()}", max_tokens=16, user=user)
|
||||
|
||||
|
||||
def _assert_budget_blocks(client: BudgetClient, key: str, *, user: str = "") -> StreamingResponse:
|
||||
|
|
@ -62,7 +63,7 @@ class TestBudgetBlocksPerLevel:
|
|||
domain=Domain.SPEND_BUDGETS,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.ANTHROPIC,),
|
||||
models=("claude-haiku-4-5",),
|
||||
models=(MODEL,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
|
|
@ -78,7 +79,7 @@ class TestBudgetBlocksPerLevel:
|
|||
domain=Domain.SPEND_BUDGETS,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.ANTHROPIC,),
|
||||
models=("claude-haiku-4-5",),
|
||||
models=(MODEL,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
|
|
@ -103,7 +104,7 @@ class TestBudgetBlocksPerLevel:
|
|||
domain=Domain.SPEND_BUDGETS,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.ANTHROPIC,),
|
||||
models=("claude-haiku-4-5",),
|
||||
models=(MODEL,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
|
|
@ -146,7 +147,7 @@ class TestBudgetBlocksPerLevel:
|
|||
domain=Domain.SPEND_BUDGETS,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.ANTHROPIC,),
|
||||
models=("claude-haiku-4-5",),
|
||||
models=(MODEL,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
|
|
@ -156,7 +157,7 @@ class TestBudgetBlocksPerLevel:
|
|||
customer = f"e2e-budget-cust-{unique_marker()}"
|
||||
client.create_customer(customer, max_budget=TINY_CAP)
|
||||
resources.defer(lambda: client.delete_customers([customer]))
|
||||
key = client.generate_key(models=["claude-haiku-4-5"])
|
||||
key = client.generate_key(models=[MODEL])
|
||||
resources.defer(lambda: client.delete_key(key))
|
||||
|
||||
_assert_budget_blocks(client, key, user=customer)
|
||||
|
|
@ -167,7 +168,7 @@ class TestBudgetBlocksPerLevel:
|
|||
domain=Domain.SPEND_BUDGETS,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.ANTHROPIC,),
|
||||
models=("claude-haiku-4-5",),
|
||||
models=(MODEL,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
|
|
@ -190,7 +191,7 @@ class TestBudgetBlocksPerLevel:
|
|||
domain=Domain.SPEND_BUDGETS,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.ANTHROPIC,),
|
||||
models=("claude-haiku-4-5",),
|
||||
models=(MODEL,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
|
|
@ -226,7 +227,7 @@ class TestKeyBudgetBlocksAcrossKeyKinds:
|
|||
domain=Domain.SPEND_BUDGETS,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.ANTHROPIC,),
|
||||
models=("claude-haiku-4-5",),
|
||||
models=(MODEL,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
|
|
@ -249,7 +250,7 @@ class TestKeyBudgetBlocksAcrossKeyKinds:
|
|||
domain=Domain.SPEND_BUDGETS,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.ANTHROPIC,),
|
||||
models=("claude-haiku-4-5",),
|
||||
models=(MODEL,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
|
|
@ -270,7 +271,7 @@ class TestKeyBudgetBlocksAcrossKeyKinds:
|
|||
domain=Domain.SPEND_BUDGETS,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.ANTHROPIC,),
|
||||
models=("claude-haiku-4-5",),
|
||||
models=(MODEL,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
|
|
|
|||
|
|
@ -28,6 +28,7 @@ from models import BudgetWindow
|
|||
|
||||
pytestmark = pytest.mark.e2e
|
||||
|
||||
MODEL = "claude-haiku-4-5"
|
||||
WINDOW_SECONDS = 30
|
||||
RESET_DEADLINE_SECONDS = 150
|
||||
TINY_CAP = 3e-6
|
||||
|
|
@ -35,7 +36,7 @@ SPEND_SETTLE_DEADLINE_SECONDS = 90
|
|||
|
||||
|
||||
def _call(client: BudgetClient, key: str):
|
||||
return client.chat(key, "claude-haiku-4-5", f"advance {unique_marker()}", max_tokens=16)
|
||||
return client.chat(key, MODEL, f"advance {unique_marker()}", max_tokens=16)
|
||||
|
||||
|
||||
def _poll_key_spend(client: BudgetClient, key: str, settled: Callable[[float], bool], problem: str) -> None:
|
||||
|
|
@ -98,7 +99,7 @@ def test_key_with_budget_duration_schedules_reset_at_creation(client: BudgetClie
|
|||
domain=Domain.SPEND_BUDGETS,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.ANTHROPIC,),
|
||||
models=("claude-haiku-4-5",),
|
||||
models=(MODEL,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
|
|
@ -124,7 +125,7 @@ def test_key_spend_blocks_at_cap(client: BudgetClient, resources: ResourceManage
|
|||
domain=Domain.SPEND_BUDGETS,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.ANTHROPIC,),
|
||||
models=("claude-haiku-4-5",),
|
||||
models=(MODEL,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
|
|
@ -169,7 +170,7 @@ def test_key_budget_reset_at_advances_after_window(client: BudgetClient, resourc
|
|||
domain=Domain.SPEND_BUDGETS,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.ANTHROPIC,),
|
||||
models=("claude-haiku-4-5",),
|
||||
models=(MODEL,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
|
|
@ -222,7 +223,7 @@ def test_multi_window_key_resets_each_window_independently(client: BudgetClient,
|
|||
domain=Domain.SPEND_BUDGETS,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.ANTHROPIC,),
|
||||
models=("claude-haiku-4-5",),
|
||||
models=(MODEL,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
|
|
@ -264,7 +265,7 @@ def test_team_member_budget_reset_at_advances(client: BudgetClient, resources: R
|
|||
domain=Domain.SPEND_BUDGETS,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.ANTHROPIC,),
|
||||
models=("claude-haiku-4-5",),
|
||||
models=(MODEL,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
|
|
|
|||
|
|
@ -12,6 +12,7 @@ from lifecycle import ResourceManager
|
|||
|
||||
pytestmark = pytest.mark.e2e
|
||||
|
||||
MODEL = "claude-haiku-4-5"
|
||||
TINY_CAP = 3e-6
|
||||
ROOMY_CAP = 100.0
|
||||
WINDOW = "30s"
|
||||
|
|
@ -19,7 +20,7 @@ RESET_DEADLINE_SECONDS = 150
|
|||
|
||||
|
||||
def _call(client: BudgetClient, key: str):
|
||||
return client.chat(key, "claude-haiku-4-5", f"reset {unique_marker()}", max_tokens=16)
|
||||
return client.chat(key, MODEL, f"reset {unique_marker()}", max_tokens=16)
|
||||
|
||||
|
||||
def _drive_to_block(client: BudgetClient, key: str) -> None:
|
||||
|
|
@ -55,7 +56,7 @@ class TestBudgetResetPerLevel:
|
|||
domain=Domain.SPEND_BUDGETS,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.ANTHROPIC,),
|
||||
models=("claude-haiku-4-5",),
|
||||
models=(MODEL,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
|
|
@ -72,7 +73,7 @@ class TestBudgetResetPerLevel:
|
|||
domain=Domain.SPEND_BUDGETS,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.ANTHROPIC,),
|
||||
models=("claude-haiku-4-5",),
|
||||
models=(MODEL,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
|
|
@ -93,7 +94,7 @@ class TestBudgetResetPerLevel:
|
|||
domain=Domain.SPEND_BUDGETS,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.ANTHROPIC,),
|
||||
models=("claude-haiku-4-5",),
|
||||
models=(MODEL,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
|
|
@ -124,7 +125,7 @@ class TestBudgetResetPerLevel:
|
|||
domain=Domain.SPEND_BUDGETS,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.ANTHROPIC,),
|
||||
models=("claude-haiku-4-5",),
|
||||
models=(MODEL,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
|
|
@ -151,7 +152,7 @@ class TestKeyBudgetResetAcrossKeyKinds:
|
|||
domain=Domain.SPEND_BUDGETS,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.ANTHROPIC,),
|
||||
models=("claude-haiku-4-5",),
|
||||
models=(MODEL,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
|
|
@ -170,7 +171,7 @@ class TestKeyBudgetResetAcrossKeyKinds:
|
|||
domain=Domain.SPEND_BUDGETS,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.ANTHROPIC,),
|
||||
models=("claude-haiku-4-5",),
|
||||
models=(MODEL,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
|
|
@ -189,7 +190,7 @@ class TestKeyBudgetResetAcrossKeyKinds:
|
|||
domain=Domain.SPEND_BUDGETS,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.ANTHROPIC,),
|
||||
models=("claude-haiku-4-5",),
|
||||
models=(MODEL,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
|
|
|
|||
|
|
@ -17,6 +17,8 @@ from lifecycle import ResourceManager
|
|||
|
||||
pytestmark = pytest.mark.e2e
|
||||
|
||||
MODEL = "claude-haiku-4-5"
|
||||
|
||||
|
||||
@pytest.mark.covers("quota_management.budget.soft.alerts_without_blocking")
|
||||
@meta(
|
||||
|
|
@ -24,7 +26,7 @@ pytestmark = pytest.mark.e2e
|
|||
domain=Domain.SPEND_BUDGETS,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.ANTHROPIC,),
|
||||
models=("claude-haiku-4-5",),
|
||||
models=(MODEL,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
|
|
@ -37,7 +39,7 @@ def test_soft_budget_does_not_block(
|
|||
|
||||
for _ in range(3):
|
||||
result = client.chat(
|
||||
key, "claude-haiku-4-5", f"hi {unique_marker()}", max_tokens=16
|
||||
key, MODEL, f"hi {unique_marker()}", max_tokens=16
|
||||
)
|
||||
assert not is_budget_block(result), (
|
||||
"soft_budget blocked a request; it must alert only, not block "
|
||||
|
|
|
|||
|
|
@ -18,13 +18,14 @@ from lifecycle import ResourceManager
|
|||
|
||||
pytestmark = pytest.mark.e2e
|
||||
|
||||
MODEL = "claude-haiku-4-5"
|
||||
TINY_BUDGET = 1e-6
|
||||
|
||||
|
||||
def _tagged_call(client: BudgetClient, key: str, tag: str):
|
||||
result = client.chat(
|
||||
key,
|
||||
"claude-haiku-4-5",
|
||||
MODEL,
|
||||
f"hi {unique_marker()}",
|
||||
tags=[tag],
|
||||
max_tokens=64,
|
||||
|
|
@ -40,7 +41,7 @@ def _tagged_call(client: BudgetClient, key: str, tag: str):
|
|||
domain=Domain.SPEND_BUDGETS,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.ANTHROPIC,),
|
||||
models=("claude-haiku-4-5",),
|
||||
models=(MODEL,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
|
|
|
|||
|
|
@ -11,6 +11,7 @@ from lifecycle import ResourceManager
|
|||
|
||||
pytestmark = pytest.mark.e2e
|
||||
|
||||
MODEL = "claude-haiku-4-5"
|
||||
MEMBER_BUDGET = 1.0 # default member budget is $50, we're testing with a smaller value
|
||||
|
||||
def _as_datetime(value: str) -> datetime:
|
||||
|
|
@ -23,7 +24,7 @@ def _as_datetime(value: str) -> datetime:
|
|||
domain=Domain.SPEND_BUDGETS,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.ANTHROPIC,),
|
||||
models=("claude-haiku-4-5",),
|
||||
models=(MODEL,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
|
|
@ -44,7 +45,7 @@ def test_team_member_budget_reset_keeps_advancing(client: BudgetClient, resource
|
|||
# the member can spend within the team while the window is live
|
||||
key = client.generate_key(team_id=team_id, user_id=user_id)
|
||||
resources.defer(lambda: client.delete_key(key))
|
||||
require_successful_call(client.chat(key, "claude-haiku-4-5", f"reset {unique_marker()}", max_tokens=16))
|
||||
require_successful_call(client.chat(key, MODEL, f"reset {unique_marker()}", max_tokens=16))
|
||||
|
||||
# once the window elapses the reset job must move budget_reset_at forward; a job
|
||||
# that skips the member's budget row (the #25109 regression) leaves it pinned at
|
||||
|
|
|
|||
|
|
@ -30,6 +30,7 @@ from models import BudgetWindow
|
|||
|
||||
pytestmark = pytest.mark.e2e
|
||||
|
||||
MODEL = "claude-haiku-4-5"
|
||||
WINDOW_SECONDS = 30
|
||||
SHORT_WINDOW = f"{WINDOW_SECONDS}s"
|
||||
LONG_WINDOW = "1d"
|
||||
|
|
@ -39,7 +40,7 @@ RESET_DEADLINE_SECONDS = 150
|
|||
|
||||
|
||||
def _call(client: BudgetClient, key: str):
|
||||
return client.chat(key, "claude-haiku-4-5", f"team-window {unique_marker()}", max_tokens=16)
|
||||
return client.chat(key, MODEL, f"team-window {unique_marker()}", max_tokens=16)
|
||||
|
||||
|
||||
def _drive_to_block(client: BudgetClient, key: str) -> StreamingResponse:
|
||||
|
|
@ -58,7 +59,7 @@ def _drive_to_block(client: BudgetClient, key: str) -> StreamingResponse:
|
|||
domain=Domain.SPEND_BUDGETS,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.ANTHROPIC,),
|
||||
models=("claude-haiku-4-5",),
|
||||
models=(MODEL,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
|
|
@ -71,7 +72,7 @@ def test_team_short_window_blocks_then_resets(client: BudgetClient, resources: R
|
|||
],
|
||||
)
|
||||
resources.defer(lambda: client.delete_team(team_id))
|
||||
key = client.generate_key(team_id=team_id, models=["claude-haiku-4-5"])
|
||||
key = client.generate_key(team_id=team_id, models=[MODEL])
|
||||
resources.defer(lambda: client.delete_key(key))
|
||||
|
||||
# 1. exhaust the tight window -> litellm returns budget_exceeded
|
||||
|
|
@ -100,7 +101,7 @@ def test_team_short_window_blocks_then_resets(client: BudgetClient, resources: R
|
|||
domain=Domain.SPEND_BUDGETS,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.ANTHROPIC,),
|
||||
models=("claude-haiku-4-5",),
|
||||
models=(MODEL,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
|
|
@ -115,7 +116,7 @@ def test_team_long_window_blocks_after_short_window_resets(client: BudgetClient,
|
|||
],
|
||||
)
|
||||
resources.defer(lambda: client.delete_team(team_id))
|
||||
key = client.generate_key(team_id=team_id, models=["claude-haiku-4-5"])
|
||||
key = client.generate_key(team_id=team_id, models=[MODEL])
|
||||
resources.defer(lambda: client.delete_key(key))
|
||||
|
||||
# 1. drive the key to being blocked, assert its blocked by budget budget_exceeded
|
||||
|
|
|
|||
|
|
@ -9,6 +9,7 @@ from lifecycle import ResourceManager
|
|||
from models import ChatBody, ChatMessage, ChatResponse, KeyGenerateBody, LiteLLMParamsBody, TeamNewBody
|
||||
from spend_e2e_client import SpendClient
|
||||
|
||||
BACKEND: Final = "openai/gpt-5.6-luna"
|
||||
INPUT_RATE: Final = 0.00004
|
||||
OUTPUT_RATE: Final = 0.00008
|
||||
|
||||
|
|
@ -38,7 +39,7 @@ def create_traffic(client: SpendClient, resources: ResourceManager) -> tuple[Tea
|
|||
model_id: Final = client.proxy.create_model(
|
||||
model,
|
||||
LiteLLMParamsBody(
|
||||
model="openai/gpt-5.6-luna",
|
||||
model=BACKEND,
|
||||
api_key="os.environ/OPENAI_API_KEY",
|
||||
api_base=None if base is None else f"{base}/v1",
|
||||
input_cost_per_token=INPUT_RATE,
|
||||
|
|
|
|||
|
|
@ -33,9 +33,16 @@ from spend_e2e_client import (
|
|||
unique_marker,
|
||||
unwrap,
|
||||
)
|
||||
from spend_reconciliation import BACKEND as TRAFFIC_BACKEND
|
||||
|
||||
pytestmark = pytest.mark.e2e
|
||||
|
||||
GEMINI_MODEL = "gemini-2.5-flash"
|
||||
CLAUDE_MODEL = "claude-haiku-4-5"
|
||||
CODEX_MODEL = "openai-responses-codex"
|
||||
EMBEDDING_MODEL = "openai-text-embedding-3-small"
|
||||
OPENAI_BACKEND = "openai/gpt-5.5"
|
||||
|
||||
|
||||
def _approx_equal(actual: float, expected: float) -> bool:
|
||||
"""Within 1% or 1e-9 absolute - spend math, not exact float identity."""
|
||||
|
|
@ -76,7 +83,7 @@ def _require_row(
|
|||
domain=Domain.SPEND_BUDGETS,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.GEMINI,),
|
||||
models=("gemini-2.5-flash",),
|
||||
models=(GEMINI_MODEL,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
|
|
@ -86,7 +93,7 @@ def test_chat_completion_writes_nonzero_spend_row(
|
|||
chat = unwrap(
|
||||
client.chat(
|
||||
scoped_key,
|
||||
"gemini-2.5-flash",
|
||||
GEMINI_MODEL,
|
||||
f"reply with one word {unique_marker()}",
|
||||
max_tokens=16,
|
||||
)
|
||||
|
|
@ -100,7 +107,7 @@ def test_chat_completion_writes_nonzero_spend_row(
|
|||
assert (row.spend or 0) > 0, f"chat row should cost > 0: {_summarize(rows)}"
|
||||
assert row.status == "success"
|
||||
assert row.cache_hit != "True", "fresh call must not be a cache hit"
|
||||
assert "gemini-2.5-flash" in (row.model or "")
|
||||
assert GEMINI_MODEL in (row.model or "")
|
||||
|
||||
prompt = row.prompt_tokens or 0
|
||||
completion = row.completion_tokens or 0
|
||||
|
|
@ -120,7 +127,7 @@ def test_chat_completion_writes_nonzero_spend_row(
|
|||
domain=Domain.SPEND_BUDGETS,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.GEMINI,),
|
||||
models=("gemini-2.5-flash",),
|
||||
models=(GEMINI_MODEL,),
|
||||
mode=Mode.STREAM,
|
||||
)
|
||||
)
|
||||
|
|
@ -129,7 +136,7 @@ def test_streaming_chat_completion_tracks_spend(
|
|||
) -> None:
|
||||
result = client.chat_stream(
|
||||
scoped_key,
|
||||
"gemini-2.5-flash",
|
||||
GEMINI_MODEL,
|
||||
f"count to three {unique_marker()}",
|
||||
max_tokens=64,
|
||||
)
|
||||
|
|
@ -157,7 +164,7 @@ def test_streaming_chat_completion_tracks_spend(
|
|||
domain=Domain.SPEND_BUDGETS,
|
||||
route=Route.MESSAGES,
|
||||
providers=(Provider.OPENAI,),
|
||||
models=("openai-responses-codex",),
|
||||
models=(CODEX_MODEL,),
|
||||
mode=Mode.STREAM,
|
||||
)
|
||||
)
|
||||
|
|
@ -178,7 +185,7 @@ def test_streaming_messages_via_responses_bridge_tracks_spend(
|
|||
"""
|
||||
result = client.messages_stream(
|
||||
scoped_key,
|
||||
"openai-responses-codex",
|
||||
CODEX_MODEL,
|
||||
f"reply with exactly one word {unique_marker()}",
|
||||
max_tokens=64,
|
||||
)
|
||||
|
|
@ -236,7 +243,7 @@ def test_streaming_messages_via_responses_bridge_tracks_spend(
|
|||
domain=Domain.SPEND_BUDGETS,
|
||||
route=Route.EMBEDDINGS,
|
||||
providers=(Provider.OPENAI,),
|
||||
models=("openai-text-embedding-3-small",),
|
||||
models=(EMBEDDING_MODEL,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
|
|
@ -246,7 +253,7 @@ def test_embedding_writes_nonzero_spend_row(
|
|||
_ = unwrap(
|
||||
client.embed(
|
||||
scoped_key,
|
||||
"openai-text-embedding-3-small",
|
||||
EMBEDDING_MODEL,
|
||||
f"vectorize this sentence {unique_marker()}",
|
||||
)
|
||||
)
|
||||
|
|
@ -268,7 +275,7 @@ def test_embedding_writes_nonzero_spend_row(
|
|||
domain=Domain.SPEND_BUDGETS,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.GEMINI,),
|
||||
models=("gemini-2.5-flash",),
|
||||
models=(GEMINI_MODEL,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
|
|
@ -280,8 +287,8 @@ def test_cache_hit_is_zero_cost_and_suffixed(
|
|||
# populated. The marker keeps each run isolated - a fixed prompt would persist
|
||||
# in the shared response cache across runs and make both calls hit (flaky).
|
||||
prompt = f"What is the capital of France? Answer in one word. {unique_marker()}"
|
||||
_ = unwrap(client.chat(scoped_key, "gemini-2.5-flash", prompt, max_tokens=16, cache=None))
|
||||
_ = unwrap(client.chat(scoped_key, "gemini-2.5-flash", prompt, max_tokens=16, cache=None))
|
||||
_ = unwrap(client.chat(scoped_key, GEMINI_MODEL, prompt, max_tokens=16, cache=None))
|
||||
_ = unwrap(client.chat(scoped_key, GEMINI_MODEL, prompt, max_tokens=16, cache=None))
|
||||
|
||||
rows = client.poll_logs_for_key(
|
||||
scoped_key,
|
||||
|
|
@ -313,7 +320,7 @@ def test_cache_hit_is_zero_cost_and_suffixed(
|
|||
domain=Domain.SPEND_BUDGETS,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.GEMINI,),
|
||||
models=("gemini-2.5-flash",),
|
||||
models=(GEMINI_MODEL,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
|
|
@ -322,7 +329,7 @@ def test_key_spend_equals_sum_of_logs(client: SpendClient, scoped_key: str) -> N
|
|||
_ = unwrap(
|
||||
client.chat(
|
||||
scoped_key,
|
||||
"gemini-2.5-flash",
|
||||
GEMINI_MODEL,
|
||||
f"say hi {unique_marker()}",
|
||||
max_tokens=16,
|
||||
)
|
||||
|
|
@ -350,7 +357,7 @@ def test_key_spend_equals_sum_of_logs(client: SpendClient, scoped_key: str) -> N
|
|||
domain=Domain.SPEND_BUDGETS,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.OPENAI,),
|
||||
models=("openai/gpt-5.6-luna",),
|
||||
models=(TRAFFIC_BACKEND,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
|
|
@ -376,7 +383,7 @@ def test_burst_of_concurrent_calls_loses_no_spend(
|
|||
domain=Domain.SPEND_BUDGETS,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.GEMINI,),
|
||||
models=("gemini-2.5-flash",),
|
||||
models=(GEMINI_MODEL,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
|
|
@ -396,7 +403,7 @@ def test_spend_logs_v2_pagination_caps_pages_and_keeps_total(
|
|||
_ = unwrap(
|
||||
client.chat(
|
||||
scoped_key,
|
||||
"gemini-2.5-flash",
|
||||
GEMINI_MODEL,
|
||||
f"page fodder {unique_marker()}",
|
||||
max_tokens=16,
|
||||
)
|
||||
|
|
@ -438,7 +445,7 @@ def test_spend_logs_v2_pagination_caps_pages_and_keeps_total(
|
|||
domain=Domain.SPEND_BUDGETS,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.GEMINI,),
|
||||
models=("gemini-2.5-flash",),
|
||||
models=(GEMINI_MODEL,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
|
|
@ -446,7 +453,7 @@ def test_request_tags_round_trip(client: SpendClient, scoped_key: str) -> None:
|
|||
tag = f"e2e-spend-{unique_marker()}"
|
||||
_ = unwrap(
|
||||
client.chat(
|
||||
scoped_key, "gemini-2.5-flash", "tagged request", tags=[tag], max_tokens=16
|
||||
scoped_key, GEMINI_MODEL, "tagged request", tags=[tag], max_tokens=16
|
||||
)
|
||||
)
|
||||
|
||||
|
|
@ -464,7 +471,7 @@ def test_request_tags_round_trip(client: SpendClient, scoped_key: str) -> None:
|
|||
domain=Domain.SPEND_BUDGETS,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.GEMINI,),
|
||||
models=("gemini-2.5-flash",),
|
||||
models=(GEMINI_MODEL,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
|
|
@ -478,7 +485,7 @@ def test_tag_spend_matches_sum_of_tagged_logs(
|
|||
_ = unwrap(
|
||||
client.chat(
|
||||
scoped_key,
|
||||
"gemini-2.5-flash",
|
||||
GEMINI_MODEL,
|
||||
f"hi {unique_marker()}",
|
||||
tags=[tag],
|
||||
max_tokens=16,
|
||||
|
|
@ -511,7 +518,7 @@ def test_tag_spend_matches_sum_of_tagged_logs(
|
|||
domain=Domain.SPEND_BUDGETS,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.GEMINI,),
|
||||
models=("gemini-2.5-flash",),
|
||||
models=(GEMINI_MODEL,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
|
|
@ -520,7 +527,7 @@ def test_end_user_spend_attributed_on_row(
|
|||
) -> None:
|
||||
customer = resources.customer(f"e2e-cust-{unique_marker()}")
|
||||
_ = unwrap(
|
||||
client.chat(scoped_key, "gemini-2.5-flash", "hi", user=customer, max_tokens=16)
|
||||
client.chat(scoped_key, GEMINI_MODEL, "hi", user=customer, max_tokens=16)
|
||||
)
|
||||
|
||||
rows = client.poll_logs_for_key(
|
||||
|
|
@ -548,7 +555,7 @@ def test_end_user_header_attributes_responses_row(
|
|||
{"authorization": f"Bearer {scoped_key}", header: customer, "x-litellm-tags": tag}
|
||||
)
|
||||
sent = client.send_responses_with_headers(
|
||||
headers, "openai-responses-codex", f"one word {unique_marker()}"
|
||||
headers, CODEX_MODEL, f"one word {unique_marker()}"
|
||||
)
|
||||
assert sent.ok, f"/v1/responses failed with {sent.status_code}: {sent.body[:300]}"
|
||||
|
||||
|
|
@ -573,7 +580,7 @@ def test_end_user_header_attributes_responses_row(
|
|||
domain=Domain.SPEND_BUDGETS,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.GEMINI, Provider.ANTHROPIC),
|
||||
models=("gemini-2.5-flash", "claude-haiku-4-5"),
|
||||
models=(GEMINI_MODEL, CLAUDE_MODEL),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
|
|
@ -587,27 +594,27 @@ def test_each_model_on_a_shared_key_gets_its_own_row(
|
|||
sibling deployment, or collapses both calls onto one request_id fails here."""
|
||||
gemini = unwrap(
|
||||
client.chat(
|
||||
scoped_key, "gemini-2.5-flash", f"one word {unique_marker()}", max_tokens=16
|
||||
scoped_key, GEMINI_MODEL, f"one word {unique_marker()}", max_tokens=16
|
||||
)
|
||||
)
|
||||
claude = unwrap(
|
||||
client.chat(
|
||||
scoped_key, "claude-haiku-4-5", f"one word {unique_marker()}", max_tokens=16
|
||||
scoped_key, CLAUDE_MODEL, f"one word {unique_marker()}", max_tokens=16
|
||||
)
|
||||
)
|
||||
|
||||
def both_models_costed(rows: list[SpendLogRow]) -> bool:
|
||||
costed = [r.model or "" for r in rows if (r.spend or 0) > 0]
|
||||
return any("gemini-2.5-flash" in m for m in costed) and any(
|
||||
"claude-haiku-4-5" in m for m in costed
|
||||
return any(GEMINI_MODEL in m for m in costed) and any(
|
||||
CLAUDE_MODEL in m for m in costed
|
||||
)
|
||||
|
||||
rows = client.poll_logs_for_key(scoped_key, min_rows=2, predicate=both_models_costed)
|
||||
gemini_row = _require_row(
|
||||
rows, lambda r: "gemini-2.5-flash" in (r.model or ""), "for the gemini call"
|
||||
rows, lambda r: GEMINI_MODEL in (r.model or ""), "for the gemini call"
|
||||
)
|
||||
claude_row = _require_row(
|
||||
rows, lambda r: "claude-haiku-4-5" in (r.model or ""), "for the claude call"
|
||||
rows, lambda r: CLAUDE_MODEL in (r.model or ""), "for the claude call"
|
||||
)
|
||||
|
||||
assert (gemini_row.spend or 0) > 0, f"gemini row should cost > 0: {_summarize(rows)}"
|
||||
|
|
@ -631,7 +638,7 @@ def test_each_model_on_a_shared_key_gets_its_own_row(
|
|||
domain=Domain.SPEND_BUDGETS,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.OPENAI,),
|
||||
models=("openai/gpt-5.5",),
|
||||
models=(OPENAI_BACKEND,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
|
|
@ -641,7 +648,7 @@ def test_failure_call_writes_failure_status_row(
|
|||
model = f"e2e-spend-failure-{unique_marker()}"
|
||||
model_id = client.proxy.create_model(
|
||||
model,
|
||||
LiteLLMParamsBody(model="openai/gpt-5.5", api_key="sk-invalid-e2e-failure-row"),
|
||||
LiteLLMParamsBody(model=OPENAI_BACKEND, api_key="sk-invalid-e2e-failure-row"),
|
||||
)
|
||||
resources.defer(lambda: client.proxy.delete_model(model_id))
|
||||
|
||||
|
|
@ -668,7 +675,7 @@ def test_failure_rows_share_normalized_error_across_provider_wording(
|
|||
carries the same stable normalized_error cluster key."""
|
||||
marker = unique_marker()
|
||||
deployments: Final = (
|
||||
(f"e2e-norm-openai-{marker}", "openai/gpt-5.5"),
|
||||
(f"e2e-norm-openai-{marker}", OPENAI_BACKEND),
|
||||
(f"e2e-norm-anthropic-{marker}", "anthropic/claude-haiku-4-5"),
|
||||
)
|
||||
for name, provider_model in deployments:
|
||||
|
|
@ -711,7 +718,7 @@ def test_pre_call_rejection_row_attributes_provider_and_model_id(
|
|||
can count it."""
|
||||
model = f"e2e-spend-precall-{unique_marker()}"
|
||||
model_id = client.proxy.create_model(
|
||||
model, LiteLLMParamsBody(model="openai/gpt-5.5", api_key="os.environ/OPENAI_API_KEY")
|
||||
model, LiteLLMParamsBody(model=OPENAI_BACKEND, api_key="os.environ/OPENAI_API_KEY")
|
||||
)
|
||||
resources.defer(lambda: client.proxy.delete_model(model_id))
|
||||
key = client.proxy.generate_key(KeyGenerateBody(models=[model], rpm_limit=1))
|
||||
|
|
@ -747,12 +754,12 @@ def test_pre_call_rejection_row_attributes_provider_and_model_id(
|
|||
domain=Domain.SPEND_BUDGETS,
|
||||
route=Route.SPEND_REPORTING,
|
||||
providers=(Provider.GEMINI,),
|
||||
models=("gemini-2.5-flash",),
|
||||
models=(GEMINI_MODEL,),
|
||||
)
|
||||
)
|
||||
def test_spend_calculate_returns_nonzero_cost(client: SpendClient) -> None:
|
||||
cost = client.calculate_spend(
|
||||
"gemini-2.5-flash", "estimate the cost of this request"
|
||||
GEMINI_MODEL, "estimate the cost of this request"
|
||||
)
|
||||
assert cost > 0, (
|
||||
"/spend/calculate returned 0 for gemini-2.5-flash; "
|
||||
|
|
@ -765,7 +772,7 @@ def test_spend_calculate_returns_nonzero_cost(client: SpendClient) -> None:
|
|||
domain=Domain.SPEND_BUDGETS,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.GEMINI,),
|
||||
models=("gemini-2.5-flash",),
|
||||
models=(GEMINI_MODEL,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
|
|
@ -779,7 +786,7 @@ def test_spend_logs_endpoint_returns_spend(
|
|||
call's nonzero spend must surface before the deadline."""
|
||||
unwrap(
|
||||
client.chat(
|
||||
scoped_key, "gemini-2.5-flash", f"spend logs {unique_marker()}", max_tokens=16
|
||||
scoped_key, GEMINI_MODEL, f"spend logs {unique_marker()}", max_tokens=16
|
||||
)
|
||||
)
|
||||
|
||||
|
|
|
|||
|
|
@ -19,7 +19,7 @@ from lifecycle import ResourceManager
|
|||
from proxy_client import Converged, await_converged
|
||||
from pydantic import BaseModel
|
||||
from spend_e2e_client import SpendClient
|
||||
from spend_reconciliation import TeamTraffic, assert_logs_match, create_traffic
|
||||
from spend_reconciliation import BACKEND, TeamTraffic, assert_logs_match, create_traffic
|
||||
|
||||
pytestmark = pytest.mark.e2e
|
||||
|
||||
|
|
@ -88,7 +88,7 @@ class TestTeamDailyActivity:
|
|||
domain=Domain.SPEND_BUDGETS,
|
||||
route=Route.SPEND_REPORTING,
|
||||
providers=(Provider.OPENAI,),
|
||||
models=("openai/gpt-5.6-luna",),
|
||||
models=(BACKEND,),
|
||||
)
|
||||
)
|
||||
def test_valid_date_range_returns_results_and_metadata(
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue