diff --git a/tests/code_coverage_tests/test_e2e_metadata.py b/tests/code_coverage_tests/test_e2e_metadata.py index 08c2984ffff..406dee8ae8a 100644 --- a/tests/code_coverage_tests/test_e2e_metadata.py +++ b/tests/code_coverage_tests/test_e2e_metadata.py @@ -272,6 +272,36 @@ class TestProviderMirrorsLitellm: assert not unknown, f"not LlmProviders values: {unknown}" +E2E_DIR: Final = Path(__file__).resolve().parents[1] / "e2e" + + +def _hand_typed_models(path: Path) -> Iterator[str]: + for node in ast.walk(ast.parse(path.read_text())): + match node: + case ast.Call(func=ast.Name(id="Subject"), keywords=keywords): + for keyword in keywords: + match keyword: + case ast.keyword(arg="models", value=ast.Tuple(elts=models)): + yield from ( + f"{path.relative_to(E2E_DIR)}:{model.lineno} {model.value!r}" + for model in models + if isinstance(model, ast.Constant) + ) + case _: + pass + case _: + pass + + +def test_a_declared_model_names_the_constant_the_test_drives() -> None: + """A model typed out in `@meta` is a second copy of the one the test calls, so + changing the call would leave the coverage report naming the old model.""" + offenders: Final = tuple( + offender for path in sorted(E2E_DIR.rglob("*.py")) for offender in _hand_typed_models(path) + ) + assert offenders == () + + class TestStepRecording: """`@step`-decorated harness helpers append to the running test's story as they execute. diff --git a/tests/e2e/quota_management/budgets/test_budget_enforcement_e2e.py b/tests/e2e/quota_management/budgets/test_budget_enforcement_e2e.py index 3e75ba1ebcc..7887ae38a74 100644 --- a/tests/e2e/quota_management/budgets/test_budget_enforcement_e2e.py +++ b/tests/e2e/quota_management/budgets/test_budget_enforcement_e2e.py @@ -24,12 +24,13 @@ from lifecycle import ResourceManager pytestmark = pytest.mark.e2e +MODEL = "claude-haiku-4-5" TINY_CAP = 3e-6 ROOMY_CAP = 100.0 def _chat(client: BudgetClient, key: str, *, user: str | None = None) -> StreamingResponse: - return client.chat(key, "claude-haiku-4-5", f"spend {unique_marker()}", max_tokens=16, user=user) + return client.chat(key, MODEL, f"spend {unique_marker()}", max_tokens=16, user=user) def _assert_budget_blocks(client: BudgetClient, key: str, *, user: str = "") -> StreamingResponse: @@ -62,7 +63,7 @@ class TestBudgetBlocksPerLevel: domain=Domain.SPEND_BUDGETS, route=Route.CHAT_COMPLETIONS, providers=(Provider.ANTHROPIC,), - models=("claude-haiku-4-5",), + models=(MODEL,), mode=Mode.NONSTREAM, ) ) @@ -78,7 +79,7 @@ class TestBudgetBlocksPerLevel: domain=Domain.SPEND_BUDGETS, route=Route.CHAT_COMPLETIONS, providers=(Provider.ANTHROPIC,), - models=("claude-haiku-4-5",), + models=(MODEL,), mode=Mode.NONSTREAM, ) ) @@ -103,7 +104,7 @@ class TestBudgetBlocksPerLevel: domain=Domain.SPEND_BUDGETS, route=Route.CHAT_COMPLETIONS, providers=(Provider.ANTHROPIC,), - models=("claude-haiku-4-5",), + models=(MODEL,), mode=Mode.NONSTREAM, ) ) @@ -146,7 +147,7 @@ class TestBudgetBlocksPerLevel: domain=Domain.SPEND_BUDGETS, route=Route.CHAT_COMPLETIONS, providers=(Provider.ANTHROPIC,), - models=("claude-haiku-4-5",), + models=(MODEL,), mode=Mode.NONSTREAM, ) ) @@ -156,7 +157,7 @@ class TestBudgetBlocksPerLevel: customer = f"e2e-budget-cust-{unique_marker()}" client.create_customer(customer, max_budget=TINY_CAP) resources.defer(lambda: client.delete_customers([customer])) - key = client.generate_key(models=["claude-haiku-4-5"]) + key = client.generate_key(models=[MODEL]) resources.defer(lambda: client.delete_key(key)) _assert_budget_blocks(client, key, user=customer) @@ -167,7 +168,7 @@ class TestBudgetBlocksPerLevel: domain=Domain.SPEND_BUDGETS, route=Route.CHAT_COMPLETIONS, providers=(Provider.ANTHROPIC,), - models=("claude-haiku-4-5",), + models=(MODEL,), mode=Mode.NONSTREAM, ) ) @@ -190,7 +191,7 @@ class TestBudgetBlocksPerLevel: domain=Domain.SPEND_BUDGETS, route=Route.CHAT_COMPLETIONS, providers=(Provider.ANTHROPIC,), - models=("claude-haiku-4-5",), + models=(MODEL,), mode=Mode.NONSTREAM, ) ) @@ -226,7 +227,7 @@ class TestKeyBudgetBlocksAcrossKeyKinds: domain=Domain.SPEND_BUDGETS, route=Route.CHAT_COMPLETIONS, providers=(Provider.ANTHROPIC,), - models=("claude-haiku-4-5",), + models=(MODEL,), mode=Mode.NONSTREAM, ) ) @@ -249,7 +250,7 @@ class TestKeyBudgetBlocksAcrossKeyKinds: domain=Domain.SPEND_BUDGETS, route=Route.CHAT_COMPLETIONS, providers=(Provider.ANTHROPIC,), - models=("claude-haiku-4-5",), + models=(MODEL,), mode=Mode.NONSTREAM, ) ) @@ -270,7 +271,7 @@ class TestKeyBudgetBlocksAcrossKeyKinds: domain=Domain.SPEND_BUDGETS, route=Route.CHAT_COMPLETIONS, providers=(Provider.ANTHROPIC,), - models=("claude-haiku-4-5",), + models=(MODEL,), mode=Mode.NONSTREAM, ) ) diff --git a/tests/e2e/quota_management/budgets/test_budget_reset_advances_e2e.py b/tests/e2e/quota_management/budgets/test_budget_reset_advances_e2e.py index 72538422874..971c2944a5f 100644 --- a/tests/e2e/quota_management/budgets/test_budget_reset_advances_e2e.py +++ b/tests/e2e/quota_management/budgets/test_budget_reset_advances_e2e.py @@ -28,6 +28,7 @@ from models import BudgetWindow pytestmark = pytest.mark.e2e +MODEL = "claude-haiku-4-5" WINDOW_SECONDS = 30 RESET_DEADLINE_SECONDS = 150 TINY_CAP = 3e-6 @@ -35,7 +36,7 @@ SPEND_SETTLE_DEADLINE_SECONDS = 90 def _call(client: BudgetClient, key: str): - return client.chat(key, "claude-haiku-4-5", f"advance {unique_marker()}", max_tokens=16) + return client.chat(key, MODEL, f"advance {unique_marker()}", max_tokens=16) def _poll_key_spend(client: BudgetClient, key: str, settled: Callable[[float], bool], problem: str) -> None: @@ -98,7 +99,7 @@ def test_key_with_budget_duration_schedules_reset_at_creation(client: BudgetClie domain=Domain.SPEND_BUDGETS, route=Route.CHAT_COMPLETIONS, providers=(Provider.ANTHROPIC,), - models=("claude-haiku-4-5",), + models=(MODEL,), mode=Mode.NONSTREAM, ) ) @@ -124,7 +125,7 @@ def test_key_spend_blocks_at_cap(client: BudgetClient, resources: ResourceManage domain=Domain.SPEND_BUDGETS, route=Route.CHAT_COMPLETIONS, providers=(Provider.ANTHROPIC,), - models=("claude-haiku-4-5",), + models=(MODEL,), mode=Mode.NONSTREAM, ) ) @@ -169,7 +170,7 @@ def test_key_budget_reset_at_advances_after_window(client: BudgetClient, resourc domain=Domain.SPEND_BUDGETS, route=Route.CHAT_COMPLETIONS, providers=(Provider.ANTHROPIC,), - models=("claude-haiku-4-5",), + models=(MODEL,), mode=Mode.NONSTREAM, ) ) @@ -222,7 +223,7 @@ def test_multi_window_key_resets_each_window_independently(client: BudgetClient, domain=Domain.SPEND_BUDGETS, route=Route.CHAT_COMPLETIONS, providers=(Provider.ANTHROPIC,), - models=("claude-haiku-4-5",), + models=(MODEL,), mode=Mode.NONSTREAM, ) ) @@ -264,7 +265,7 @@ def test_team_member_budget_reset_at_advances(client: BudgetClient, resources: R domain=Domain.SPEND_BUDGETS, route=Route.CHAT_COMPLETIONS, providers=(Provider.ANTHROPIC,), - models=("claude-haiku-4-5",), + models=(MODEL,), mode=Mode.NONSTREAM, ) ) diff --git a/tests/e2e/quota_management/budgets/test_budget_reset_e2e.py b/tests/e2e/quota_management/budgets/test_budget_reset_e2e.py index d3984cb9712..ff9185fa9b2 100644 --- a/tests/e2e/quota_management/budgets/test_budget_reset_e2e.py +++ b/tests/e2e/quota_management/budgets/test_budget_reset_e2e.py @@ -12,6 +12,7 @@ from lifecycle import ResourceManager pytestmark = pytest.mark.e2e +MODEL = "claude-haiku-4-5" TINY_CAP = 3e-6 ROOMY_CAP = 100.0 WINDOW = "30s" @@ -19,7 +20,7 @@ RESET_DEADLINE_SECONDS = 150 def _call(client: BudgetClient, key: str): - return client.chat(key, "claude-haiku-4-5", f"reset {unique_marker()}", max_tokens=16) + return client.chat(key, MODEL, f"reset {unique_marker()}", max_tokens=16) def _drive_to_block(client: BudgetClient, key: str) -> None: @@ -55,7 +56,7 @@ class TestBudgetResetPerLevel: domain=Domain.SPEND_BUDGETS, route=Route.CHAT_COMPLETIONS, providers=(Provider.ANTHROPIC,), - models=("claude-haiku-4-5",), + models=(MODEL,), mode=Mode.NONSTREAM, ) ) @@ -72,7 +73,7 @@ class TestBudgetResetPerLevel: domain=Domain.SPEND_BUDGETS, route=Route.CHAT_COMPLETIONS, providers=(Provider.ANTHROPIC,), - models=("claude-haiku-4-5",), + models=(MODEL,), mode=Mode.NONSTREAM, ) ) @@ -93,7 +94,7 @@ class TestBudgetResetPerLevel: domain=Domain.SPEND_BUDGETS, route=Route.CHAT_COMPLETIONS, providers=(Provider.ANTHROPIC,), - models=("claude-haiku-4-5",), + models=(MODEL,), mode=Mode.NONSTREAM, ) ) @@ -124,7 +125,7 @@ class TestBudgetResetPerLevel: domain=Domain.SPEND_BUDGETS, route=Route.CHAT_COMPLETIONS, providers=(Provider.ANTHROPIC,), - models=("claude-haiku-4-5",), + models=(MODEL,), mode=Mode.NONSTREAM, ) ) @@ -151,7 +152,7 @@ class TestKeyBudgetResetAcrossKeyKinds: domain=Domain.SPEND_BUDGETS, route=Route.CHAT_COMPLETIONS, providers=(Provider.ANTHROPIC,), - models=("claude-haiku-4-5",), + models=(MODEL,), mode=Mode.NONSTREAM, ) ) @@ -170,7 +171,7 @@ class TestKeyBudgetResetAcrossKeyKinds: domain=Domain.SPEND_BUDGETS, route=Route.CHAT_COMPLETIONS, providers=(Provider.ANTHROPIC,), - models=("claude-haiku-4-5",), + models=(MODEL,), mode=Mode.NONSTREAM, ) ) @@ -189,7 +190,7 @@ class TestKeyBudgetResetAcrossKeyKinds: domain=Domain.SPEND_BUDGETS, route=Route.CHAT_COMPLETIONS, providers=(Provider.ANTHROPIC,), - models=("claude-haiku-4-5",), + models=(MODEL,), mode=Mode.NONSTREAM, ) ) diff --git a/tests/e2e/quota_management/budgets/test_soft_budget_e2e.py b/tests/e2e/quota_management/budgets/test_soft_budget_e2e.py index c1a840b8b1e..ecba8ad71f0 100644 --- a/tests/e2e/quota_management/budgets/test_soft_budget_e2e.py +++ b/tests/e2e/quota_management/budgets/test_soft_budget_e2e.py @@ -17,6 +17,8 @@ from lifecycle import ResourceManager pytestmark = pytest.mark.e2e +MODEL = "claude-haiku-4-5" + @pytest.mark.covers("quota_management.budget.soft.alerts_without_blocking") @meta( @@ -24,7 +26,7 @@ pytestmark = pytest.mark.e2e domain=Domain.SPEND_BUDGETS, route=Route.CHAT_COMPLETIONS, providers=(Provider.ANTHROPIC,), - models=("claude-haiku-4-5",), + models=(MODEL,), mode=Mode.NONSTREAM, ) ) @@ -37,7 +39,7 @@ def test_soft_budget_does_not_block( for _ in range(3): result = client.chat( - key, "claude-haiku-4-5", f"hi {unique_marker()}", max_tokens=16 + key, MODEL, f"hi {unique_marker()}", max_tokens=16 ) assert not is_budget_block(result), ( "soft_budget blocked a request; it must alert only, not block " diff --git a/tests/e2e/quota_management/budgets/test_tag_budget_e2e.py b/tests/e2e/quota_management/budgets/test_tag_budget_e2e.py index cb7a90458f1..4022dfc847d 100644 --- a/tests/e2e/quota_management/budgets/test_tag_budget_e2e.py +++ b/tests/e2e/quota_management/budgets/test_tag_budget_e2e.py @@ -18,13 +18,14 @@ from lifecycle import ResourceManager pytestmark = pytest.mark.e2e +MODEL = "claude-haiku-4-5" TINY_BUDGET = 1e-6 def _tagged_call(client: BudgetClient, key: str, tag: str): result = client.chat( key, - "claude-haiku-4-5", + MODEL, f"hi {unique_marker()}", tags=[tag], max_tokens=64, @@ -40,7 +41,7 @@ def _tagged_call(client: BudgetClient, key: str, tag: str): domain=Domain.SPEND_BUDGETS, route=Route.CHAT_COMPLETIONS, providers=(Provider.ANTHROPIC,), - models=("claude-haiku-4-5",), + models=(MODEL,), mode=Mode.NONSTREAM, ) ) diff --git a/tests/e2e/quota_management/budgets/test_team_member_budget_reset_e2e.py b/tests/e2e/quota_management/budgets/test_team_member_budget_reset_e2e.py index d39c7e1af62..552554023e6 100644 --- a/tests/e2e/quota_management/budgets/test_team_member_budget_reset_e2e.py +++ b/tests/e2e/quota_management/budgets/test_team_member_budget_reset_e2e.py @@ -11,6 +11,7 @@ from lifecycle import ResourceManager pytestmark = pytest.mark.e2e +MODEL = "claude-haiku-4-5" MEMBER_BUDGET = 1.0 # default member budget is $50, we're testing with a smaller value def _as_datetime(value: str) -> datetime: @@ -23,7 +24,7 @@ def _as_datetime(value: str) -> datetime: domain=Domain.SPEND_BUDGETS, route=Route.CHAT_COMPLETIONS, providers=(Provider.ANTHROPIC,), - models=("claude-haiku-4-5",), + models=(MODEL,), mode=Mode.NONSTREAM, ) ) @@ -44,7 +45,7 @@ def test_team_member_budget_reset_keeps_advancing(client: BudgetClient, resource # the member can spend within the team while the window is live key = client.generate_key(team_id=team_id, user_id=user_id) resources.defer(lambda: client.delete_key(key)) - require_successful_call(client.chat(key, "claude-haiku-4-5", f"reset {unique_marker()}", max_tokens=16)) + require_successful_call(client.chat(key, MODEL, f"reset {unique_marker()}", max_tokens=16)) # once the window elapses the reset job must move budget_reset_at forward; a job # that skips the member's budget row (the #25109 regression) leaves it pinned at diff --git a/tests/e2e/quota_management/budgets/test_team_multi_window_budget_e2e.py b/tests/e2e/quota_management/budgets/test_team_multi_window_budget_e2e.py index 0994c28ce16..d45a599f5ee 100644 --- a/tests/e2e/quota_management/budgets/test_team_multi_window_budget_e2e.py +++ b/tests/e2e/quota_management/budgets/test_team_multi_window_budget_e2e.py @@ -30,6 +30,7 @@ from models import BudgetWindow pytestmark = pytest.mark.e2e +MODEL = "claude-haiku-4-5" WINDOW_SECONDS = 30 SHORT_WINDOW = f"{WINDOW_SECONDS}s" LONG_WINDOW = "1d" @@ -39,7 +40,7 @@ RESET_DEADLINE_SECONDS = 150 def _call(client: BudgetClient, key: str): - return client.chat(key, "claude-haiku-4-5", f"team-window {unique_marker()}", max_tokens=16) + return client.chat(key, MODEL, f"team-window {unique_marker()}", max_tokens=16) def _drive_to_block(client: BudgetClient, key: str) -> StreamingResponse: @@ -58,7 +59,7 @@ def _drive_to_block(client: BudgetClient, key: str) -> StreamingResponse: domain=Domain.SPEND_BUDGETS, route=Route.CHAT_COMPLETIONS, providers=(Provider.ANTHROPIC,), - models=("claude-haiku-4-5",), + models=(MODEL,), mode=Mode.NONSTREAM, ) ) @@ -71,7 +72,7 @@ def test_team_short_window_blocks_then_resets(client: BudgetClient, resources: R ], ) resources.defer(lambda: client.delete_team(team_id)) - key = client.generate_key(team_id=team_id, models=["claude-haiku-4-5"]) + key = client.generate_key(team_id=team_id, models=[MODEL]) resources.defer(lambda: client.delete_key(key)) # 1. exhaust the tight window -> litellm returns budget_exceeded @@ -100,7 +101,7 @@ def test_team_short_window_blocks_then_resets(client: BudgetClient, resources: R domain=Domain.SPEND_BUDGETS, route=Route.CHAT_COMPLETIONS, providers=(Provider.ANTHROPIC,), - models=("claude-haiku-4-5",), + models=(MODEL,), mode=Mode.NONSTREAM, ) ) @@ -115,7 +116,7 @@ def test_team_long_window_blocks_after_short_window_resets(client: BudgetClient, ], ) resources.defer(lambda: client.delete_team(team_id)) - key = client.generate_key(team_id=team_id, models=["claude-haiku-4-5"]) + key = client.generate_key(team_id=team_id, models=[MODEL]) resources.defer(lambda: client.delete_key(key)) # 1. drive the key to being blocked, assert its blocked by budget budget_exceeded diff --git a/tests/e2e/quota_management/spend_tracking/spend_reconciliation.py b/tests/e2e/quota_management/spend_tracking/spend_reconciliation.py index 26809874aed..f313325dbda 100644 --- a/tests/e2e/quota_management/spend_tracking/spend_reconciliation.py +++ b/tests/e2e/quota_management/spend_tracking/spend_reconciliation.py @@ -9,6 +9,7 @@ from lifecycle import ResourceManager from models import ChatBody, ChatMessage, ChatResponse, KeyGenerateBody, LiteLLMParamsBody, TeamNewBody from spend_e2e_client import SpendClient +BACKEND: Final = "openai/gpt-5.6-luna" INPUT_RATE: Final = 0.00004 OUTPUT_RATE: Final = 0.00008 @@ -38,7 +39,7 @@ def create_traffic(client: SpendClient, resources: ResourceManager) -> tuple[Tea model_id: Final = client.proxy.create_model( model, LiteLLMParamsBody( - model="openai/gpt-5.6-luna", + model=BACKEND, api_key="os.environ/OPENAI_API_KEY", api_base=None if base is None else f"{base}/v1", input_cost_per_token=INPUT_RATE, diff --git a/tests/e2e/quota_management/spend_tracking/test_spend_tracking_e2e.py b/tests/e2e/quota_management/spend_tracking/test_spend_tracking_e2e.py index 43caed32425..ab3b6cd6389 100644 --- a/tests/e2e/quota_management/spend_tracking/test_spend_tracking_e2e.py +++ b/tests/e2e/quota_management/spend_tracking/test_spend_tracking_e2e.py @@ -33,9 +33,16 @@ from spend_e2e_client import ( unique_marker, unwrap, ) +from spend_reconciliation import BACKEND as TRAFFIC_BACKEND pytestmark = pytest.mark.e2e +GEMINI_MODEL = "gemini-2.5-flash" +CLAUDE_MODEL = "claude-haiku-4-5" +CODEX_MODEL = "openai-responses-codex" +EMBEDDING_MODEL = "openai-text-embedding-3-small" +OPENAI_BACKEND = "openai/gpt-5.5" + def _approx_equal(actual: float, expected: float) -> bool: """Within 1% or 1e-9 absolute - spend math, not exact float identity.""" @@ -76,7 +83,7 @@ def _require_row( domain=Domain.SPEND_BUDGETS, route=Route.CHAT_COMPLETIONS, providers=(Provider.GEMINI,), - models=("gemini-2.5-flash",), + models=(GEMINI_MODEL,), mode=Mode.NONSTREAM, ) ) @@ -86,7 +93,7 @@ def test_chat_completion_writes_nonzero_spend_row( chat = unwrap( client.chat( scoped_key, - "gemini-2.5-flash", + GEMINI_MODEL, f"reply with one word {unique_marker()}", max_tokens=16, ) @@ -100,7 +107,7 @@ def test_chat_completion_writes_nonzero_spend_row( assert (row.spend or 0) > 0, f"chat row should cost > 0: {_summarize(rows)}" assert row.status == "success" assert row.cache_hit != "True", "fresh call must not be a cache hit" - assert "gemini-2.5-flash" in (row.model or "") + assert GEMINI_MODEL in (row.model or "") prompt = row.prompt_tokens or 0 completion = row.completion_tokens or 0 @@ -120,7 +127,7 @@ def test_chat_completion_writes_nonzero_spend_row( domain=Domain.SPEND_BUDGETS, route=Route.CHAT_COMPLETIONS, providers=(Provider.GEMINI,), - models=("gemini-2.5-flash",), + models=(GEMINI_MODEL,), mode=Mode.STREAM, ) ) @@ -129,7 +136,7 @@ def test_streaming_chat_completion_tracks_spend( ) -> None: result = client.chat_stream( scoped_key, - "gemini-2.5-flash", + GEMINI_MODEL, f"count to three {unique_marker()}", max_tokens=64, ) @@ -157,7 +164,7 @@ def test_streaming_chat_completion_tracks_spend( domain=Domain.SPEND_BUDGETS, route=Route.MESSAGES, providers=(Provider.OPENAI,), - models=("openai-responses-codex",), + models=(CODEX_MODEL,), mode=Mode.STREAM, ) ) @@ -178,7 +185,7 @@ def test_streaming_messages_via_responses_bridge_tracks_spend( """ result = client.messages_stream( scoped_key, - "openai-responses-codex", + CODEX_MODEL, f"reply with exactly one word {unique_marker()}", max_tokens=64, ) @@ -236,7 +243,7 @@ def test_streaming_messages_via_responses_bridge_tracks_spend( domain=Domain.SPEND_BUDGETS, route=Route.EMBEDDINGS, providers=(Provider.OPENAI,), - models=("openai-text-embedding-3-small",), + models=(EMBEDDING_MODEL,), mode=Mode.NONSTREAM, ) ) @@ -246,7 +253,7 @@ def test_embedding_writes_nonzero_spend_row( _ = unwrap( client.embed( scoped_key, - "openai-text-embedding-3-small", + EMBEDDING_MODEL, f"vectorize this sentence {unique_marker()}", ) ) @@ -268,7 +275,7 @@ def test_embedding_writes_nonzero_spend_row( domain=Domain.SPEND_BUDGETS, route=Route.CHAT_COMPLETIONS, providers=(Provider.GEMINI,), - models=("gemini-2.5-flash",), + models=(GEMINI_MODEL,), mode=Mode.NONSTREAM, ) ) @@ -280,8 +287,8 @@ def test_cache_hit_is_zero_cost_and_suffixed( # populated. The marker keeps each run isolated - a fixed prompt would persist # in the shared response cache across runs and make both calls hit (flaky). prompt = f"What is the capital of France? Answer in one word. {unique_marker()}" - _ = unwrap(client.chat(scoped_key, "gemini-2.5-flash", prompt, max_tokens=16, cache=None)) - _ = unwrap(client.chat(scoped_key, "gemini-2.5-flash", prompt, max_tokens=16, cache=None)) + _ = unwrap(client.chat(scoped_key, GEMINI_MODEL, prompt, max_tokens=16, cache=None)) + _ = unwrap(client.chat(scoped_key, GEMINI_MODEL, prompt, max_tokens=16, cache=None)) rows = client.poll_logs_for_key( scoped_key, @@ -313,7 +320,7 @@ def test_cache_hit_is_zero_cost_and_suffixed( domain=Domain.SPEND_BUDGETS, route=Route.CHAT_COMPLETIONS, providers=(Provider.GEMINI,), - models=("gemini-2.5-flash",), + models=(GEMINI_MODEL,), mode=Mode.NONSTREAM, ) ) @@ -322,7 +329,7 @@ def test_key_spend_equals_sum_of_logs(client: SpendClient, scoped_key: str) -> N _ = unwrap( client.chat( scoped_key, - "gemini-2.5-flash", + GEMINI_MODEL, f"say hi {unique_marker()}", max_tokens=16, ) @@ -350,7 +357,7 @@ def test_key_spend_equals_sum_of_logs(client: SpendClient, scoped_key: str) -> N domain=Domain.SPEND_BUDGETS, route=Route.CHAT_COMPLETIONS, providers=(Provider.OPENAI,), - models=("openai/gpt-5.6-luna",), + models=(TRAFFIC_BACKEND,), mode=Mode.NONSTREAM, ) ) @@ -376,7 +383,7 @@ def test_burst_of_concurrent_calls_loses_no_spend( domain=Domain.SPEND_BUDGETS, route=Route.CHAT_COMPLETIONS, providers=(Provider.GEMINI,), - models=("gemini-2.5-flash",), + models=(GEMINI_MODEL,), mode=Mode.NONSTREAM, ) ) @@ -396,7 +403,7 @@ def test_spend_logs_v2_pagination_caps_pages_and_keeps_total( _ = unwrap( client.chat( scoped_key, - "gemini-2.5-flash", + GEMINI_MODEL, f"page fodder {unique_marker()}", max_tokens=16, ) @@ -438,7 +445,7 @@ def test_spend_logs_v2_pagination_caps_pages_and_keeps_total( domain=Domain.SPEND_BUDGETS, route=Route.CHAT_COMPLETIONS, providers=(Provider.GEMINI,), - models=("gemini-2.5-flash",), + models=(GEMINI_MODEL,), mode=Mode.NONSTREAM, ) ) @@ -446,7 +453,7 @@ def test_request_tags_round_trip(client: SpendClient, scoped_key: str) -> None: tag = f"e2e-spend-{unique_marker()}" _ = unwrap( client.chat( - scoped_key, "gemini-2.5-flash", "tagged request", tags=[tag], max_tokens=16 + scoped_key, GEMINI_MODEL, "tagged request", tags=[tag], max_tokens=16 ) ) @@ -464,7 +471,7 @@ def test_request_tags_round_trip(client: SpendClient, scoped_key: str) -> None: domain=Domain.SPEND_BUDGETS, route=Route.CHAT_COMPLETIONS, providers=(Provider.GEMINI,), - models=("gemini-2.5-flash",), + models=(GEMINI_MODEL,), mode=Mode.NONSTREAM, ) ) @@ -478,7 +485,7 @@ def test_tag_spend_matches_sum_of_tagged_logs( _ = unwrap( client.chat( scoped_key, - "gemini-2.5-flash", + GEMINI_MODEL, f"hi {unique_marker()}", tags=[tag], max_tokens=16, @@ -511,7 +518,7 @@ def test_tag_spend_matches_sum_of_tagged_logs( domain=Domain.SPEND_BUDGETS, route=Route.CHAT_COMPLETIONS, providers=(Provider.GEMINI,), - models=("gemini-2.5-flash",), + models=(GEMINI_MODEL,), mode=Mode.NONSTREAM, ) ) @@ -520,7 +527,7 @@ def test_end_user_spend_attributed_on_row( ) -> None: customer = resources.customer(f"e2e-cust-{unique_marker()}") _ = unwrap( - client.chat(scoped_key, "gemini-2.5-flash", "hi", user=customer, max_tokens=16) + client.chat(scoped_key, GEMINI_MODEL, "hi", user=customer, max_tokens=16) ) rows = client.poll_logs_for_key( @@ -548,7 +555,7 @@ def test_end_user_header_attributes_responses_row( {"authorization": f"Bearer {scoped_key}", header: customer, "x-litellm-tags": tag} ) sent = client.send_responses_with_headers( - headers, "openai-responses-codex", f"one word {unique_marker()}" + headers, CODEX_MODEL, f"one word {unique_marker()}" ) assert sent.ok, f"/v1/responses failed with {sent.status_code}: {sent.body[:300]}" @@ -573,7 +580,7 @@ def test_end_user_header_attributes_responses_row( domain=Domain.SPEND_BUDGETS, route=Route.CHAT_COMPLETIONS, providers=(Provider.GEMINI, Provider.ANTHROPIC), - models=("gemini-2.5-flash", "claude-haiku-4-5"), + models=(GEMINI_MODEL, CLAUDE_MODEL), mode=Mode.NONSTREAM, ) ) @@ -587,27 +594,27 @@ def test_each_model_on_a_shared_key_gets_its_own_row( sibling deployment, or collapses both calls onto one request_id fails here.""" gemini = unwrap( client.chat( - scoped_key, "gemini-2.5-flash", f"one word {unique_marker()}", max_tokens=16 + scoped_key, GEMINI_MODEL, f"one word {unique_marker()}", max_tokens=16 ) ) claude = unwrap( client.chat( - scoped_key, "claude-haiku-4-5", f"one word {unique_marker()}", max_tokens=16 + scoped_key, CLAUDE_MODEL, f"one word {unique_marker()}", max_tokens=16 ) ) def both_models_costed(rows: list[SpendLogRow]) -> bool: costed = [r.model or "" for r in rows if (r.spend or 0) > 0] - return any("gemini-2.5-flash" in m for m in costed) and any( - "claude-haiku-4-5" in m for m in costed + return any(GEMINI_MODEL in m for m in costed) and any( + CLAUDE_MODEL in m for m in costed ) rows = client.poll_logs_for_key(scoped_key, min_rows=2, predicate=both_models_costed) gemini_row = _require_row( - rows, lambda r: "gemini-2.5-flash" in (r.model or ""), "for the gemini call" + rows, lambda r: GEMINI_MODEL in (r.model or ""), "for the gemini call" ) claude_row = _require_row( - rows, lambda r: "claude-haiku-4-5" in (r.model or ""), "for the claude call" + rows, lambda r: CLAUDE_MODEL in (r.model or ""), "for the claude call" ) assert (gemini_row.spend or 0) > 0, f"gemini row should cost > 0: {_summarize(rows)}" @@ -631,7 +638,7 @@ def test_each_model_on_a_shared_key_gets_its_own_row( domain=Domain.SPEND_BUDGETS, route=Route.CHAT_COMPLETIONS, providers=(Provider.OPENAI,), - models=("openai/gpt-5.5",), + models=(OPENAI_BACKEND,), mode=Mode.NONSTREAM, ) ) @@ -641,7 +648,7 @@ def test_failure_call_writes_failure_status_row( model = f"e2e-spend-failure-{unique_marker()}" model_id = client.proxy.create_model( model, - LiteLLMParamsBody(model="openai/gpt-5.5", api_key="sk-invalid-e2e-failure-row"), + LiteLLMParamsBody(model=OPENAI_BACKEND, api_key="sk-invalid-e2e-failure-row"), ) resources.defer(lambda: client.proxy.delete_model(model_id)) @@ -668,7 +675,7 @@ def test_failure_rows_share_normalized_error_across_provider_wording( carries the same stable normalized_error cluster key.""" marker = unique_marker() deployments: Final = ( - (f"e2e-norm-openai-{marker}", "openai/gpt-5.5"), + (f"e2e-norm-openai-{marker}", OPENAI_BACKEND), (f"e2e-norm-anthropic-{marker}", "anthropic/claude-haiku-4-5"), ) for name, provider_model in deployments: @@ -711,7 +718,7 @@ def test_pre_call_rejection_row_attributes_provider_and_model_id( can count it.""" model = f"e2e-spend-precall-{unique_marker()}" model_id = client.proxy.create_model( - model, LiteLLMParamsBody(model="openai/gpt-5.5", api_key="os.environ/OPENAI_API_KEY") + model, LiteLLMParamsBody(model=OPENAI_BACKEND, api_key="os.environ/OPENAI_API_KEY") ) resources.defer(lambda: client.proxy.delete_model(model_id)) key = client.proxy.generate_key(KeyGenerateBody(models=[model], rpm_limit=1)) @@ -747,12 +754,12 @@ def test_pre_call_rejection_row_attributes_provider_and_model_id( domain=Domain.SPEND_BUDGETS, route=Route.SPEND_REPORTING, providers=(Provider.GEMINI,), - models=("gemini-2.5-flash",), + models=(GEMINI_MODEL,), ) ) def test_spend_calculate_returns_nonzero_cost(client: SpendClient) -> None: cost = client.calculate_spend( - "gemini-2.5-flash", "estimate the cost of this request" + GEMINI_MODEL, "estimate the cost of this request" ) assert cost > 0, ( "/spend/calculate returned 0 for gemini-2.5-flash; " @@ -765,7 +772,7 @@ def test_spend_calculate_returns_nonzero_cost(client: SpendClient) -> None: domain=Domain.SPEND_BUDGETS, route=Route.CHAT_COMPLETIONS, providers=(Provider.GEMINI,), - models=("gemini-2.5-flash",), + models=(GEMINI_MODEL,), mode=Mode.NONSTREAM, ) ) @@ -779,7 +786,7 @@ def test_spend_logs_endpoint_returns_spend( call's nonzero spend must surface before the deadline.""" unwrap( client.chat( - scoped_key, "gemini-2.5-flash", f"spend logs {unique_marker()}", max_tokens=16 + scoped_key, GEMINI_MODEL, f"spend logs {unique_marker()}", max_tokens=16 ) ) diff --git a/tests/e2e/quota_management/spend_tracking/test_team_daily_activity_e2e.py b/tests/e2e/quota_management/spend_tracking/test_team_daily_activity_e2e.py index 30e34a6475d..c86b55dc990 100644 --- a/tests/e2e/quota_management/spend_tracking/test_team_daily_activity_e2e.py +++ b/tests/e2e/quota_management/spend_tracking/test_team_daily_activity_e2e.py @@ -19,7 +19,7 @@ from lifecycle import ResourceManager from proxy_client import Converged, await_converged from pydantic import BaseModel from spend_e2e_client import SpendClient -from spend_reconciliation import TeamTraffic, assert_logs_match, create_traffic +from spend_reconciliation import BACKEND, TeamTraffic, assert_logs_match, create_traffic pytestmark = pytest.mark.e2e @@ -88,7 +88,7 @@ class TestTeamDailyActivity: domain=Domain.SPEND_BUDGETS, route=Route.SPEND_REPORTING, providers=(Provider.OPENAI,), - models=("openai/gpt-5.6-luna",), + models=(BACKEND,), ) ) def test_valid_date_range_returns_results_and_metadata(