diff --git a/tests/e2e/coverage_registry/mgmt.yaml b/tests/e2e/coverage_registry/mgmt.yaml index 2a0fc5c9f29..5d80ab5e15c 100644 --- a/tests/e2e/coverage_registry/mgmt.yaml +++ b/tests/e2e/coverage_registry/mgmt.yaml @@ -54,6 +54,7 @@ - {id: mgmt.access_group.info.happy_path, module: mgmt, tier: P1, surface: api, assertions: [happy_path], source: "model_access_group_management_endpoints.py:600", rationale: "Access group membership query"} - {id: mgmt.mcp_server.register.happy_path, module: mgmt, tier: P1, surface: api, assertions: [happy_path], source: "mcp_management_endpoints.py:880", rationale: "MCP server registration"} - {id: mgmt.mcp_server.approve.persists, module: mgmt, tier: P1, surface: api, assertions: [persists], source: "mcp_management_endpoints.py:1200", rationale: "Admin approval persists"} +- {id: mgmt.budget.update.accepts_model_max_budget, module: mgmt, tier: P1, surface: api, assertions: [accepts_model_max_budget], source: "budget_management_endpoints.py:173", fail_before_fix: proven, rationale: "Per-model caps must be settable on an existing budget; model ids routinely carry dots and hyphens and the route must accept both"} - {id: mgmt.budget.update.persists, module: mgmt, tier: P1, surface: api, assertions: [persists], source: "budget_management_endpoints.py:155", rationale: "Limit changes apply"} - {id: mgmt.budget.delete.persists, module: mgmt, tier: P1, surface: api, assertions: [persists], source: "budget_management_endpoints.py:280", rationale: "Clears limits"} - {id: mgmt.budget.list.happy_path, module: mgmt, tier: P1, surface: api, assertions: [happy_path], source: "budget_management_endpoints.py:215", rationale: "Budget enumeration"} diff --git a/tests/e2e/coverage_registry/quota_management.yaml b/tests/e2e/coverage_registry/quota_management.yaml index eb620395c46..2dfa7adddea 100644 --- a/tests/e2e/coverage_registry/quota_management.yaml +++ b/tests/e2e/coverage_registry/quota_management.yaml @@ -18,6 +18,7 @@ - {id: quota_management.budget.team_member.blocks_over_limit, module: quota_management, tier: P1, behavior: budget, variant: team_member, assertions: [blocks_over_limit], exercised_on: [chat_completions], source: "proxy/auth/auth_checks.py", rationale: "A member's per-team budget blocks independently of the team budget"} - {id: quota_management.budget.team_member.isolates_per_member, module: quota_management, tier: P1, behavior: budget, variant: team_member, assertions: [isolates_per_member], exercised_on: [chat_completions], source: "proxy/auth/auth_checks.py", rationale: "One team member's exhausted per-team budget does not block a different member on the same team"} - {id: quota_management.budget.tag.blocks_over_limit, module: quota_management, tier: P1, behavior: budget, variant: tag, assertions: [blocks_over_limit], exercised_on: [chat_completions], source: "router_strategy/budget_limiter.py", rationale: "Proxy-level tag budgets block tagged requests at the cap"} +- {id: quota_management.budget.end_user_model_max.blocks_over_limit, module: quota_management, tier: P1, behavior: budget, variant: end_user_model_max, assertions: [blocks_over_limit], exercised_on: [chat_completions], source: "budget_management_endpoints.py", fail_before_fix: proven, rationale: "A per-model rpm_limit on an end-user budget is accepted and stored but never enforced; only key-attached budgets honour it"} - {id: quota_management.budget.model_max.isolates_per_model, module: quota_management, tier: P1, behavior: budget, variant: model_max, assertions: [isolates_per_model], exercised_on: [chat_completions], source: "proxy/hooks/model_max_budget_limiter.py", rationale: "model_max_budget caps one model without touching a sibling's budget"} - {id: quota_management.budget.soft.alerts_without_blocking, module: quota_management, tier: P1, behavior: budget, variant: soft, assertions: [alerts_without_blocking], exercised_on: [chat_completions], source: "proxy/auth/auth_checks.py", rationale: "soft_budget alerts but never blocks traffic"} - {id: quota_management.budget.key.resets_after_window, module: quota_management, tier: P1, behavior: budget, variant: key, assertions: [resets_after_window], exercised_on: [chat_completions], source: "proxy/common_utils/reset_budget_job.py", rationale: "budget_duration zeroes key spend after the window; a blocked key serves again"} diff --git a/tests/e2e/llm_translation/test_passthrough_e2e.py b/tests/e2e/llm_translation/test_passthrough_e2e.py index ed5c657d23e..b57164df9bb 100644 --- a/tests/e2e/llm_translation/test_passthrough_e2e.py +++ b/tests/e2e/llm_translation/test_passthrough_e2e.py @@ -66,6 +66,36 @@ def test_gemini_passthrough_nonstreaming_logs_cost( assert tag in (row.request_tags or []), f"tags not logged: {row.request_tags}" +@pytest.mark.skip(reason="stage red: product gap, native passthrough returns no x-litellm-response-cost or x-ratelimit-* headers") +def test_gemini_passthrough_returns_the_same_header_contract_as_the_managed_route( + client: PassthroughClient, scoped_key: str +) -> None: + """Native /gemini/ passthrough must return the same operational headers as + /chat/completions: x-litellm-response-cost so the call reconciles against + spend, and x-ratelimit-* so a client can pace itself. It returns neither + today, which makes native traffic invisible to the same tooling. + """ + result = client.gemini_generate( + scoped_key, "gemini-2.5-flash", f"Say hello in one word. {unique_marker()}" + ) + require_successful_call(result) + + assert result.call_id, "passthrough must stamp x-litellm-call-id" + assert result.response_cost is not None, ( + "passthrough generateContent returned no x-litellm-response-cost header, so a " + "native call cannot be reconciled against spend the way /chat/completions can" + ) + assert result.response_cost > 0, ( + f"x-litellm-response-cost must be a real cost, got {result.response_cost}" + ) + + pacing = tuple(name for name in result.headers if name.startswith("x-ratelimit-")) + assert pacing, ( + "passthrough generateContent returned no x-ratelimit-* headers, so a client " + f"cannot pace itself; headers present were {sorted(result.headers)}" + ) + + def test_gemini_passthrough_streaming_logs_cost( client: PassthroughClient, scoped_key: str ) -> None: diff --git a/tests/e2e/management/test_budget_customer_user_org_e2e.py b/tests/e2e/management/test_budget_customer_user_org_e2e.py index 12372bb7cc1..ae78555de78 100644 --- a/tests/e2e/management/test_budget_customer_user_org_e2e.py +++ b/tests/e2e/management/test_budget_customer_user_org_e2e.py @@ -22,7 +22,7 @@ import pytest from pydantic import BaseModel, Field, RootModel from e2e_config import unique_marker -from e2e_http import NoBody, Success, UnauthorizedError, UnknownApiError, unwrap +from e2e_http import NoBody, Success, UnauthorizedError, UnknownApiError, is_ok, unwrap from lifecycle import ResourceManager from management_client import ManagementClient from models import KeyGenerateBody, OrgInfoParams, OrgNewBody, UserNewBody @@ -55,9 +55,15 @@ class BudgetNewResponse(BaseModel): budget_id: str +class ModelBudgetEntry(BaseModel): + budget_limit: float + time_period: str + + class BudgetUpdateBody(BaseModel): budget_id: str - max_budget: float + max_budget: float | None = None + model_max_budget: dict[str, ModelBudgetEntry] | None = None class BudgetInfoBody(BaseModel): @@ -68,6 +74,7 @@ class BudgetRow(BaseModel): budget_id: str | None = None max_budget: float | None = None soft_budget: float | None = None + model_max_budget: dict[str, ModelBudgetEntry] | None = None class BudgetInfoResponse(RootModel[list[BudgetRow]]): @@ -118,6 +125,17 @@ def _budget_rows(client: ManagementClient, budget_id: str) -> tuple[BudgetRow, . ) +def _find_model_budget( + client: ManagementClient, budget_id: str, model_name: str +) -> BudgetRow | None: + row = next( + (r for r in _budget_rows(client, budget_id) if r.budget_id == budget_id), None + ) + if row is None or not row.model_max_budget or model_name not in row.model_max_budget: + return None + return row + + def _budget_list_ids(client: ManagementClient) -> tuple[str, ...]: return tuple( row.budget_id @@ -150,6 +168,53 @@ class TestBudgetManagement: f"/budget/list never included the created budget {budget_id}", ) + @pytest.mark.skip(reason="stage red: product gap, /budget/update 500s on any model_max_budget (prisma Json arg + unquoted GraphQL interpolation)") + @pytest.mark.covers("mgmt.budget.update.accepts_model_max_budget") + def test_update_accepts_per_model_budgets_including_punctuated_names( + self, client: ManagementClient, resources: ResourceManager + ) -> None: + """/budget/update must accept per-model caps on an existing budget. + + model_max_budget keys are model ids, which routinely carry dots and + hyphens (glm-5.2). Both a plain and a punctuated id are exercised so a + failure says whether per-model budgets break outright or only for + punctuated ids. + """ + for model_name in ("gpt4o", "glm-5.2"): + self._assert_model_budget_round_trips(client, resources, model_name) + + @staticmethod + def _assert_model_budget_round_trips( + client: ManagementClient, resources: ResourceManager, model_name: str + ) -> None: + budget_id = _create_budget( + client, resources, BudgetNewBody(max_budget=_INITIAL_MAX_BUDGET) + ) + + result = client.proxy.transport.post( + "/budget/update", + headers=client.proxy.transport.master, + json=BudgetUpdateBody( + budget_id=budget_id, + model_max_budget={ + model_name: ModelBudgetEntry(budget_limit=5.0, time_period="1d") + }, + ), + response_type=NoBody, + ) + + assert is_ok(result), ( + f"/budget/update rejected a per-model budget for {model_name!r}: {result}; " + f"a customer cannot cap spend per model on an existing budget" + ) + + _ = _poll( + client, + lambda: _find_model_budget(client, budget_id, model_name), + f"/budget/info never reported a model_max_budget entry for {model_name!r} " + f"on budget {budget_id}", + ) + @pytest.mark.covers("mgmt.budget.update.persists") def test_update_max_budget_persists_to_budget_info( self, client: ManagementClient, resources: ResourceManager diff --git a/tests/e2e/models.py b/tests/e2e/models.py index bef199cdd16..da4b27d6010 100644 --- a/tests/e2e/models.py +++ b/tests/e2e/models.py @@ -18,6 +18,8 @@ from pydantic import BaseModel, ConfigDict, RootModel, model_validator class ModelBudgetEntry(BaseModel): budget_limit: float time_period: str + rpm_limit: int | None = None + tpm_limit: int | None = None class BudgetWindow(BaseModel): diff --git a/tests/e2e/quota_management/budgets/budget_client.py b/tests/e2e/quota_management/budgets/budget_client.py index 83e8f27b597..5b9253928af 100644 --- a/tests/e2e/quota_management/budgets/budget_client.py +++ b/tests/e2e/quota_management/budgets/budget_client.py @@ -61,7 +61,8 @@ class UserDeleteBody(BaseModel): class CustomerNewBody(BaseModel): user_id: str - max_budget: float + max_budget: float | None = None + budget_id: str | None = None class OrgNewBody(BaseModel): @@ -151,9 +152,10 @@ class TagDeleteBody(BaseModel): class BudgetNewBody(BaseModel): - max_budget: float + max_budget: float | None = None soft_budget: float | None = None budget_duration: str | None = None + model_max_budget: dict[str, ModelBudgetEntry] | None = None class BudgetNewResponse(BaseModel): @@ -326,11 +328,19 @@ class BudgetClient: # ---- customer / end-user ------------------------------------------- - def create_customer(self, customer_id: str, *, max_budget: float) -> str: + def create_customer( + self, + customer_id: str, + *, + max_budget: float | None = None, + budget_id: str | None = None, + ) -> str: resp = self.proxy.transport.send( "/customer/new", headers=self.proxy.transport.master, - json=CustomerNewBody(user_id=customer_id, max_budget=max_budget), + json=CustomerNewBody( + user_id=customer_id, max_budget=max_budget, budget_id=budget_id + ), ) assert resp.ok, resp.body return customer_id @@ -509,9 +519,10 @@ class BudgetClient: def create_budget( self, *, - max_budget: float, + max_budget: float | None = None, soft_budget: float | None = None, budget_duration: str | None = None, + model_max_budget: dict[str, ModelBudgetEntry] | None = None, ) -> str: return unwrap( self.proxy.transport.post( @@ -521,6 +532,7 @@ class BudgetClient: max_budget=max_budget, soft_budget=soft_budget, budget_duration=budget_duration, + model_max_budget=model_max_budget, ), response_type=BudgetNewResponse, ) diff --git a/tests/e2e/quota_management/budgets/test_model_max_budget_e2e.py b/tests/e2e/quota_management/budgets/test_model_max_budget_e2e.py index 4d0df2c35ea..3f7264365ef 100644 --- a/tests/e2e/quota_management/budgets/test_model_max_budget_e2e.py +++ b/tests/e2e/quota_management/budgets/test_model_max_budget_e2e.py @@ -14,6 +14,7 @@ from budget_client import BudgetClient, is_budget_block, model_budget from e2e_config import unique_marker from e2e_http import require_successful_call from lifecycle import ResourceManager +from models import ModelBudgetEntry pytestmark = pytest.mark.e2e @@ -56,3 +57,49 @@ def test_model_max_budget_isolates_per_model( f"{FREE_MODEL} was blocked by {CAPPED_MODEL}'s budget; per-model caps not isolated" ) require_successful_call(other) + + +@pytest.mark.skip(reason="stage red: product gap, end-user model_max_budget rpm_limit is stored but never enforced") +@pytest.mark.covers("quota_management.budget.end_user_model_max.blocks_over_limit") +def test_end_user_model_max_budget_enforces_per_model_rpm( + client: BudgetClient, resources: ResourceManager +) -> None: + """A per-model rpm_limit on an end-user budget must actually throttle. + + model_max_budget takes an rpm_limit alongside the spend cap, letting a + customer hold one end user to a slow rate without limiting the shared key. + The budget hangs off the end user, not the key; the key-attached shape + already works, so this pins the end-user gap. + """ + budget_id = client.create_budget( + model_max_budget={ + FREE_MODEL: ModelBudgetEntry( + budget_limit=1000.0, time_period="1d", rpm_limit=1 + ) + } + ) + resources.defer(lambda: client.delete_budget(budget_id)) + + customer = f"e2e-mmb-cust-{unique_marker()}" + _ = client.create_customer(customer, budget_id=budget_id) + resources.defer(lambda: client.delete_customers([customer])) + + key = client.generate_key() + resources.defer(lambda: client.delete_key(key)) + + statuses = tuple( + client.chat( + key, FREE_MODEL, f"hi {unique_marker()}", max_tokens=8, user=customer + ).status_code + for _ in range(3) + ) + + assert statuses[0] == 200, ( + f"the first call under an rpm_limit of 1 should succeed, got {statuses[0]}" + ) + assert 429 in statuses[1:], ( + f"an end-user budget with model_max_budget rpm_limit=1 did not throttle: " + f"three calls returned {statuses}. The limit is accepted and stored by " + f"/budget/new but never enforced for end-user budgets, so a customer " + f"cannot rate-limit an individual end user on a shared key" + )