diff --git a/tests/e2e/llm_translation/test_passthrough_e2e.py b/tests/e2e/llm_translation/test_passthrough_e2e.py index 016f3b7be41..081e88e493b 100644 --- a/tests/e2e/llm_translation/test_passthrough_e2e.py +++ b/tests/e2e/llm_translation/test_passthrough_e2e.py @@ -69,13 +69,10 @@ def test_gemini_passthrough_nonstreaming_logs_cost( def test_gemini_passthrough_returns_the_same_header_contract_as_the_managed_route( client: PassthroughClient, scoped_key: str ) -> None: - """A native passthrough call must still be costable and pace-able by the client. - - Customers front provider-native traffic through /gemini/ and read the same - operational headers they get on /chat/completions: the response cost, so the - call reconciles against spend, and the x-ratelimit-* pacing headers, so a - client knows how much budget it has left. The passthrough route currently - returns neither, which makes native traffic invisible to the same tooling. + """Native /gemini/ passthrough must return the same operational headers as + /chat/completions: x-litellm-response-cost so the call reconciles against + spend, and x-ratelimit-* so a client can pace itself. It returns neither + today, which makes native traffic invisible to the same tooling. """ result = client.gemini_generate( scoped_key, "gemini-2.5-flash", f"Say hello in one word. {unique_marker()}" diff --git a/tests/e2e/management/test_budget_customer_user_org_e2e.py b/tests/e2e/management/test_budget_customer_user_org_e2e.py index 607c7094654..cc5423affeb 100644 --- a/tests/e2e/management/test_budget_customer_user_org_e2e.py +++ b/tests/e2e/management/test_budget_customer_user_org_e2e.py @@ -125,6 +125,17 @@ def _budget_rows(client: ManagementClient, budget_id: str) -> tuple[BudgetRow, . ) +def _find_model_budget( + client: ManagementClient, budget_id: str, model_name: str +) -> BudgetRow | None: + row = next( + (r for r in _budget_rows(client, budget_id) if r.budget_id == budget_id), None + ) + if row is None or not row.model_max_budget or model_name not in row.model_max_budget: + return None + return row + + def _budget_list_ids(client: ManagementClient) -> tuple[str, ...]: return tuple( row.budget_id @@ -161,51 +172,47 @@ class TestBudgetManagement: def test_update_accepts_per_model_budgets_including_punctuated_names( self, client: ManagementClient, resources: ResourceManager ) -> None: - """Per-model caps must be settable on an existing budget. + """/budget/update must accept per-model caps on an existing budget. - `model_max_budget` is how a customer caps spend per model on a shared - budget, and model ids routinely carry dots and hyphens (`glm-5.2`). The - route has to accept both, and the plain name is included so a failure - says whether per-model budgets are broken outright or only for + model_max_budget keys are model ids, which routinely carry dots and + hyphens (glm-5.2). Both a plain and a punctuated id are exercised so a + failure says whether per-model budgets break outright or only for punctuated ids. """ for model_name in ("gpt4o", "glm-5.2"): - budget_id = _create_budget( - client, resources, BudgetNewBody(max_budget=_INITIAL_MAX_BUDGET) - ) + self._assert_model_budget_round_trips(client, resources, model_name) - result = client.proxy.transport.post( - "/budget/update", - headers=client.proxy.transport.master, - json=BudgetUpdateBody( - budget_id=budget_id, - model_max_budget={ - model_name: ModelBudgetEntry(budget_limit=5.0, time_period="1d") - }, - ), - response_type=NoBody, - ) + @staticmethod + def _assert_model_budget_round_trips( + client: ManagementClient, resources: ResourceManager, model_name: str + ) -> None: + budget_id = _create_budget( + client, resources, BudgetNewBody(max_budget=_INITIAL_MAX_BUDGET) + ) - assert is_ok(result), ( - f"/budget/update rejected a per-model budget for {model_name!r}: {result}; " - f"a customer cannot cap spend per model on an existing budget" - ) + result = client.proxy.transport.post( + "/budget/update", + headers=client.proxy.transport.master, + json=BudgetUpdateBody( + budget_id=budget_id, + model_max_budget={ + model_name: ModelBudgetEntry(budget_limit=5.0, time_period="1d") + }, + ), + response_type=NoBody, + ) - def has_model_budget() -> BudgetRow | None: - row = next( - (r for r in _budget_rows(client, budget_id) if r.budget_id == budget_id), - None, - ) - if row is None or not row.model_max_budget: - return None - return row if model_name in row.model_max_budget else None + assert is_ok(result), ( + f"/budget/update rejected a per-model budget for {model_name!r}: {result}; " + f"a customer cannot cap spend per model on an existing budget" + ) - _ = _poll( - client, - has_model_budget, - f"/budget/info never reported a model_max_budget entry for {model_name!r} " - f"on budget {budget_id}", - ) + _ = _poll( + client, + lambda: _find_model_budget(client, budget_id, model_name), + f"/budget/info never reported a model_max_budget entry for {model_name!r} " + f"on budget {budget_id}", + ) @pytest.mark.covers("mgmt.budget.update.persists") def test_update_max_budget_persists_to_budget_info( diff --git a/tests/e2e/quota_management/budgets/test_model_max_budget_e2e.py b/tests/e2e/quota_management/budgets/test_model_max_budget_e2e.py index eb08da62d6b..f202bb570b6 100644 --- a/tests/e2e/quota_management/budgets/test_model_max_budget_e2e.py +++ b/tests/e2e/quota_management/budgets/test_model_max_budget_e2e.py @@ -11,10 +11,10 @@ import time import pytest from budget_client import BudgetClient, is_budget_block, model_budget -from models import ModelBudgetEntry from e2e_config import unique_marker from e2e_http import require_successful_call from lifecycle import ResourceManager +from models import ModelBudgetEntry pytestmark = pytest.mark.e2e @@ -63,13 +63,12 @@ def test_model_max_budget_isolates_per_model( def test_end_user_model_max_budget_enforces_per_model_rpm( client: BudgetClient, resources: ResourceManager ) -> None: - """A per-model rpm_limit on an end user's budget has to actually throttle. + """A per-model rpm_limit on an end-user budget must actually throttle. - `model_max_budget` accepts an `rpm_limit` alongside the spend cap, and a - customer uses it to hold one end user to a slow rate on an expensive model - without limiting the shared key everyone else runs through. The budget is - attached to the end user rather than to the key, which is the case that - matters here: the same shape already works when the budget hangs off a key. + model_max_budget takes an rpm_limit alongside the spend cap, letting a + customer hold one end user to a slow rate without limiting the shared key. + The budget hangs off the end user, not the key; the key-attached shape + already works, so this pins the end-user gap. """ budget_id = client.create_budget( model_max_budget={